The Scope and Limitations of Extant Research into ChatGPT as a Tool for Patient Education: Systematic Review

preprint OA: closed
📄 Open PDF Full text JSON View at publisher

Abstract

Background Chat Generative Pre-Trained Transformer (ChatGPT), a large language model (LLM) developed by OpenAI, has been extensively studied and embraced by medical researchers since its public release in November 2022. In addition to its benefits in generating summaries and predictive diagnostics, it has been proposed as a patient education tool. Existing research cites ChatGPT’s potential to increase information availability and accessibility, improve the efficiency of clinical practice, and its high-quality responses to clinical questions as reasons to consider its adoption. Objective We assessed literature from PubMed on the quality and consistency of ChatGPT responses across medical specialties, evaluated the comprehensiveness of current research, and identified areas for future research. Methods We searched PubMed for published articles evaluating the consistency, reliability, and ethics of ChatGPT in patient education. Following the PRISMA guideline, we conducted a systematic literature review of the 567 retrieved records. After title, abstract, and full-text screens, 123 relevant records were included and synthesized in this review. Results We found a lack of consensus among ChatGPT studies. The model accuracy suffers from infrequent updates and generation of misleading information (hallucinations), and it lacks knowledge of current clinical guidelines. The consistency of the model falls short due to its sensitivity to prompt design and fine-tuning through user interaction, making ChatGPT research results almost impossible to peer review and validate. Relying on ChatGPT for clinical information risks spreading misinformation, disrupting trust in the medical system, and disobeying the principles of patient-centered care. This also shifts the focus of patient education from shared decision-making and information-building to information-giving, deviating from the objectives of Health Communication and Health Literacy outlined in Healthy People 2030. Conclusions We caution against the acceptance of ChatGPT as a patient education tool and encourage future efforts to incorporate community perspectives in AI research.
Full text 72,634 characters · extracted from preprint-html · click to expand
The Scope and Limitations of Extant Research into ChatGPT as a Tool for Patient Education: Systematic Review | medRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-P4HH5NV'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search The Scope and Limitations of Extant Research into ChatGPT as a Tool for Patient Education: Systematic Review View ORCID Profile Reid Dale , View ORCID Profile Maggie Cheng , View ORCID Profile Katharine Casselman Pines , View ORCID Profile Maria Elizabeth Currie doi: https://doi.org/10.1101/2025.05.20.25328009 Reid Dale 1 Department of Cardiothoracic Surgery, Stanford School of Medicine , Stanford, CA, USA 2 Stanford Cardiovascular Institute , Stanford, CA, USA PhD Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Reid Dale Maggie Cheng 1 Department of Cardiothoracic Surgery, Stanford School of Medicine , Stanford, CA, USA BA BS Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Maggie Cheng Katharine Casselman Pines 1 Department of Cardiothoracic Surgery, Stanford School of Medicine , Stanford, CA, USA MPH Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Katharine Casselman Pines Maria Elizabeth Currie 1 Department of Cardiothoracic Surgery, Stanford School of Medicine , Stanford, CA, USA 2 Stanford Cardiovascular Institute , Stanford, CA, USA MD PhD Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Maria Elizabeth Currie For correspondence: mecurrie{at}stanford.edu Abstract Full Text Info/History Metrics Data/Code Preview PDF Abstract Background Chat Generative Pre-Trained Transformer (ChatGPT), a large language model (LLM) developed by OpenAI, has been extensively studied and embraced by medical researchers since its public release in November 2022. In addition to its benefits in generating summaries and predictive diagnostics, it has been proposed as a patient education tool. Existing research cites ChatGPT’s potential to increase information availability and accessibility, improve the efficiency of clinical practice, and its high-quality responses to clinical questions as reasons to consider its adoption. Objective We assessed literature from PubMed on the quality and consistency of ChatGPT responses across medical specialties, evaluated the comprehensiveness of current research, and identified areas for future research. Methods We searched PubMed for published articles evaluating the consistency, reliability, and ethics of ChatGPT in patient education. Following the PRISMA guideline, we conducted a systematic literature review of the 567 retrieved records. After title, abstract, and full-text screens, 123 relevant records were included and synthesized in this review. Results We found a lack of consensus among ChatGPT studies. The model accuracy suffers from infrequent updates and generation of misleading information (hallucinations), and it lacks knowledge of current clinical guidelines. The consistency of the model falls short due to its sensitivity to prompt design and fine-tuning through user interaction, making ChatGPT research results almost impossible to peer review and validate. Relying on ChatGPT for clinical information risks spreading misinformation, disrupting trust in the medical system, and disobeying the principles of patient-centered care. This also shifts the focus of patient education from shared decision-making and information-building to information-giving, deviating from the objectives of Health Communication and Health Literacy outlined in Healthy People 2030. Conclusions We caution against the acceptance of ChatGPT as a patient education tool and encourage future efforts to incorporate community perspectives in AI research. Introduction Chat Generative Pre-Trained Transformer (ChatGPT), initially developed by OpenAI in 2020, is a large language model (LLM) based on GPT-3.5. It was trained with a general information database and can recognize patterns in languages, extract information, and generate coherent and human-like texts. 1 – 4 GPT models have undergone several revisions, from GPT-1 in 2018 to GPT-4 in 2023, with progressively increased parameters and training dataset sizes. Although some studies noted significant improvement in accuracy as the model developed (especially comparing GPT 3.5 to GPT 4), others concluded that limitations in previous versions of GPT models still apply to the newest edition now. 4 – 7 Although ChatGPT shows excellent performance in sourcing and summarizing information from its training data and could be useful in improving the quality of writing, it is fundamentally a language-based model that lacks expertise in specific subject areas. 3 , 8 – 14 When applied to medical settings as a tool for individualized information sourcing or patient education, its inconsistency and lack of up-to-date medical knowledge raise technical and ethical concerns. Despite the presence of BioGPT, a model trained with biomedical research texts that presumably should be more accurate and rigorous in biomedical settings, the limitations of ChatGPT still apply. 4 , 15 At the current stage, although artificial intelligence (AI) technologies have been incorporated into some medical settings such as designing radiology treatment, the incorporation of ChatGPT in offering clinical information to patients remains controversial. 16 Much of the medical community and published literature support the utilization of ChatGPT in aiding clinical decision-making and as a valuable information source for patients despite recognizing the many flaws this technology carries, citing the model’s near-instantaneous responses, accessibility, affordability, breadth of knowledge, capacity for a personalized experience, and minimization of bias with ongoing efforts to improve training dataset. 4 , 12 , 16 – 41 However, opposite voices also exist, bringing to light the inconsistency, inaccuracy, privacy, and legal concerns regarding the incorporation of ChatGPT into clinical settings. 2 , 7 , 42 – 58 In this study, we reviewed literature from PubMed on the design, expandability, consistency, and accuracy of ChatGPT responses across medical specialties, evaluated current research, and identified areas for future research. Despite recognizing ChatGPT’s merits in increasing efficiency and the accessibility of information, it is imperative to critically evaluate the potential concerns in embracing this change too quickly. Methods Search Strategy This comprehensive review was conducted according to the Preferred Reporting Items for Systematic Reviews and Meta-Analyses (PRISMA) guidelines. We searched PubMed for published articles (including preprints) on ChatGPT and its applications in providing medical information to patients. Keywords and search terms included “ChatGPT AND consistency”, “ChatGPT AND ethics”, “ChatGPT AND health literacy”, “ChatGPT AND patient education”, and “ChatGPT AND reliability.” Literature search concluded on January 7, 2024. Inclusion and Exclusion Criteria We included published articles (including preprints) on the use of ChatGPT in clinical information dissemination as well as ethics discussion of the model. Exclusion criteria included 1) Letter to Editors (article type), 2) Comments (article type), 3) Records discussing role of ChatGPT in professional/medical education, 4) ChatGPT in scientific writing and publishing, 5) ChatGPT-assisted writing, 6) ChatGPT in diagnostics and imaging, 7) ChatGPT in translation, 8) non-English records, and 9) other irrelevant records. We did not restrict the search and inclusion criteria only to records from the United States. Articles from other countries were included as long as the full text was available and accessible in English. Results Summary of Search Results and Record Screening Process A total of 567 records were retrieved from PubMed. These represent a pooled result from all search terms: “ChatGPT AND consistency” (n = 97), “ChatGPT AND ethics” (n = 245), “ChatGPT AND health literacy” (n = 26), “ChatGPT AND patient education” (n = 64), and “ChatGPT AND reliability” (n = 136). 87 records were removed for duplication. A title screen (n = 481) was conducted next, and 274 articles that fitted the exclusion criteria were excluded. Article abstracts (n = 207) were retrieved and screened; 2 were removed for lack of access to the abstract, 14 were removed due to article type, and 56 were excluded for outside of the scope of this review. Lastly, we did a full-text article screen (n = 135); 7 articles were removed for lack of access, 3 were removed for being non-English records, and 1 were removed for being outside the scope of this review. At the end, 123 articles were included in this review. Figure 1 shows this process in a PRISMA diagram. Download figure Open in new tab Figure 1. PRISMA 2020 flow diagram for new systematic reviews which included searches of databases and registers only ChatGPT Design The development of ChatGPT and other ChatGPT-like AI chatbots have the potential to revolutionize some facets of healthcare delivery. Although these models were not developed specifically for a healthcare context, such models could lead to more effective delegation of clinical responsibilities, increase accessibility to health information, and improve clinical decision-making. 3 , 59 Beyond sourcing information, ChatGPT also summarizes and presents information concisely, saving readers time from visiting multiple different sites to extrapolate information themselves. 60 , 61 It also can translate, making information available and accessible to patients of all language backgrounds, although the accuracy and reliability of ChatGPT in non-English contexts is yet to be fully evaluated. 4 , 16 The readability of ChatGPT-generated content has also shown to be above the level that is appropriate to use in educational materials for the general public. 22 , 50 , 61 – 69 ChatGPT is a pre-trained model with the ability to fine-tune through user interactions. With delays in updating the training dataset and the resources needed for such training, ChatGPT’s response lags behind current clinical guidelines and provides outdated information. GPT 3.5 was updated in January 2022, and GPT 4-Turbo was trained in April 2023. 70 Updates in clinical guidelines after those dates would be neglected in GPT responses, potentially leading to inaccurate information. 71 ChatGPT is also subject to temperature control and top_p sampling, which determines the flexibility of the model in selecting the immediate next word as well as the set of phrases to use when generating responses. Lower temperature and top parameters give the model less flexibility in creatively generating responses and higher parameters give the model a higher degree of randomness and unpredictability. As both parameters are non-zero in GPT models, it is almost impossible to replicate an exact same response due to the irreducibly probabilistic nature of an LLM’s response. 72 , 73 Discussion Expandability and Privacy Concerns of ChatGPT Existing studies identified several concerns regarding the usage of ChatGPT in a health education context. First, the language model operates like a “black box”, not explaining its reasoning or judgments. 16 , 74 , 75 From the user’s perspective, we could only control the input but would not know what to expect as the output. The process that ChatGPT uses to generate responses is unclear, even to its developers from OpenAI. 76 This takes away the shared nature of decision-making that is supposed to happen in a clinical setting, deviating from the ideals of patient-centered medicine that the medical communities strive to emphasize. Without information on how the model arrives at its recommendations, providers are ill-prepared should patients have any questions about ChatGPT’s responses. Such an inability to properly explain or trace back the logic of AI may instill distrust between the patient and the provider, painting a picture of AI as more knowledgeable while the provider as inferior. Without information on how machine learning models work to generate a response, providers do not have adequate evidence to trust or question the validity of the conclusion, putting them in a dilemma. Their behaviors and clinical judgments may be affected by their willingness to minimize information discrepancy and maintain trust with patients, incentivizing them to side with information from the AI that they may not have full confidence in. In situations where seeking medical information from AI becomes a habit or norm, and patient is faced with conflicting information from the human physician and the AI, the patient would be unsure of how to proceed. Nov et al. gave early evidence of this trend, showing that participants could only correctly distinguish between AI vs. physician-generated responses 49-85.7% of the times, and they tended to trust ChatGPT responses when it comes to low-risk health questions. 77 When AI-generated information is used and acted upon in clinical settings, the decisions patients make may not be considered truly informed due to the unknown clinical reasoning process of ChatGPT. Second, medico-legal considerations are widely mentioned in the published literature. 35 , 47 , 57 , 60 , 78 To date, we lack regulations and accountability measures that govern the proper utilization of AI-based chatbots in medical settings. 79 – 83 Developers are not required to publicly reveal when their models may fail. 16 In the case of a medical incident, the liability of parties remains unclear. 60 , 75 Therefore, when the liability risks are truly at stake, it is unclear whether developers would be as willing to release the AI chatbots to the general public as they do now. Third, ChatGPT usage brings privacy concerns. 50 , 57 , 74 Nasr et al. revealed possibilities to extract ChatGPT’s training data, raising concerns about confidentiality. 84 With its ability to widely source and retain information, efforts should be made to limit the possibility of linking de-identified information to personally identifiable information to ensure confidentiality if ChatGPT were to be incorporated in clinical settings. 11 Consistency of ChatGPT Responses With ChatGPT’s sensitivity to prompt design and its constantly evolving nature, mixed results exist regarding the consistency of ChatGPT responses. Through assessing ChatGPT’s knowledge on five hepato-pancreatico-biliary–related conditions by inputting designed questions into the model three times, Walker et al. showed an internal consistency of 100% of the algorithm. 13 When two experts compared ChatGPT’s answers with existing guidelines in the study, the interrater agreement showed complete agreement and that the information and diagnoses provided by ChatGPT are consistent with the national guidelines. 13 However, Johnson et al. found statistically significant differences in the accuracy of ChatGPT responses to their question prompts when the same set of questions were asked to ChatGPT only a few days apart. 85 Johnson et al. explained such differences as ChatGPT’s ability to learn and fine-tune through user interactions. 85 However, this can also be interpreted as a lack of consistency in ChatGPT responses. The inconsistency of ChatGPT was further supported by Salas et al. who found that ChatGPT provided different answers to the same prompt within a short time frame. 86 The finetuning of ChatGPT through user feedback makes it susceptible to manipulation. ChatGPT cannot necessarily distinguish between right or wrong information and lacks factual knowledge in some areas. 87 , 88 When ChatGPT is corrected by the users, the corrections could potentially enter the training data of ChatGPT and affect model performance in the future. 87 If the user feedback were to be incorrect, the model could include inaccurate information or biased views in its training set, allowing the spread of misinformation iteratively. A recent breakdown of the ChatGPT system raised concerns about the consistency of the model. Deviating from its usual behavior, users reported that the chatbot gave incorrect, irrelevant, and inappropriate answers to their questions and sometimes responded with the same phrase repeatedly without providing a comprehensible sentence. Although this system breakdown was quickly fixed by the developers, OpenAI could not explain why the model failed and admitted that the model behavior can be unpredictable. 76 Such unpredictability should not be tolerated if the model were to be used for health education or information collection, where each piece of information could inform one’s health decision-making and affect trust in the medical system. With increasingly positive comments from the medical community on ChatGPT’s use in clinics, patients may see those AI chatbots as increasingly reliable, therefore developing dependence on the use of such algorithms and foregoing the knowledge they may gain from interacting with human practitioners. 30 , 31 Moreover, ChatGPT is an LLM that detects patterns in language, and it is susceptible to prompt design variations. Jang and Lukasiewicz showed that both ChatGPT (based on GPT-3) and GPT-4 can be self-contradictory, generating different answers when the input questions convey the same meaning. 89 The same team also demonstrated that GPT-3 often violates logical consistency, and the frequency of such mistakes is not negligible. These limitations were observed in previous versions of GPT models like GPT-2 but remained uncorrected despite using a larger training dataset for subsequent GPT models. Accuracy and Comprehensiveness of ChatGPT Responses When it comes to the accuracy and comprehensiveness of answers, ChatGPT’s responses vary by field. Many studies commented highly on the accuracy of ChatGPT in providing medical information. 85 , 90 – 92 However, events of AI hallucination sometimes appear, providing information or sources that do not exist. 8 , 13 , 35 , 46 , 87 , 93 , 94 The accuracy of ChatGPT reported by current literature range form 40-50% in some fields (e.g. hepatocellular carcinoma, retinal diseases, liver cancer) to greater than 90% in others (e.g. type 2 diabetes, thoracic surgery, robotic-assisted radical prostatectomy, otolaryngology). 11 , 25 , 26 , 41 , 95 – 98 Assessing ChatGPT’s knowledge across 17 specialties, Johnson et al. found that 8.3% of all answers generated were completely incorrect, and many more were acceptable but not comprehensive. Even for a new GPT model that was trained on biomedical texts, its accuracy was only 51%. 85 The consistency and accuracy of ChatGPT responses are expected to decrease with increased length of conversation, as ChatGPT sometimes struggles to recognize or maintain context in prolonged conversations. 4 , 99 Since ChatGPT is a pre-trained model with limited fine-tuning during use, its accuracy depends on the quality of the training data as well as the frequency of updates. Biases carried in training data will be represented, if not increased, in use. 2 , 8 , 47 , 74 , 78 , 100 Alarmingly, Acerbi and Stubbersfield revealed that ChatGPT-3 showed bias towards content that is gender-stereotypical, negative, and threat-related, warning the public of the limitation of such model. 101 Koranteng et al. found similar gender stereotypes in ChatGPT responses as well as its tendency to associate negative terms with African American names. 102 Also, inadequate updates of the model lead to outdated and misleading information. 1 , 71 , 103 Studying ChatGPT’s response to misconceptions about vaccinations, Deiana et al. found various inaccuracies in responses from both ChatGPT GPT 3.5 and GPT 4, and the chatbot seemed to selectively disregard the benefits of vaccines. 1 Similarly, many other studies across specialties also pointed out that the information ChatGPT provides is not up to date and can be fabricated. 71 , 91 , 104 , 105 In addition to generating inaccurate answers to the user’s questions, ChatGPT also frequently fails to provide accurate or accessible references, if at all. Eleven of the included articles pointed out that ChatGPT either did not provide references despite prompting or generated fabricated references that did not exist. 93 , 105 – 114 This inability to cite its sources exist regardless of the information accuracy provided by ChatGPT. Alessandri-Bonetti et al. showed that among all requests, ChatGPT failed to provide references 33% of the time. 106 When references were provided, 36% were shown to be irrelevant or nonexistent. 106 McGowan et al. revealed that only 6% of the references provided by ChatGPT in their study was accurate. 105 ChatGPT also has limited ability to assess the quality of such citations, potentially biasing towards sources that are present in higher quantity in its training data but not necessarily of high quality or accuracy. Potential Impacts on Patient-Physician Relationship Although only a few of the current studies clearly advice the incorporation of ChatGPT in clinical setting for information retrieval purposes, much of the literature supports the idea of using ChatGPT as a supplementary tool to aid health professionals in promoting patient education and health literacy. 11 , 60 , 115 – 122 Whereas the efficiency and relative accuracy of ChatGPT maybe appealing, we should not overlook the challenges ChatGPT faces as well as the patient-provider relationship that is vital in building health literacy. 123 , 124 Health literacy must not only advance people’s knowledge and ability to obtain and understand health information, but also empower them to actively participate in the health decision-making process. This includes discerning the quality of information, communicating their concerns, asking questions, and self-manage their conditions. 16 , 123 – 126 Putting the accuracy and consistency concerns about ChatGPT aside, merely providing information to patients is not an effective way to improve health literacy and can be dangerously biased. Just as the black-box nature of GPT models goes against the key principles of patient-centered medicine and informed consent, receiving health advice based on the limited patient-provided information to ChatGPT carries the same risks. Modern medicine emphasizes both evidence-based practice and patient-centered care. Recommending the utilization of ChatGPT for obtaining medical information may dangerously reduce people into numbers and data points and potentially ignore their beliefs and values. 16 On the contrary, healthcare professionals play a key role in incorporating such considerations through the establishment of a trusting patient-provider relationship. 16 Such a relationship should not be a hierarchical one of information-giving and receiving or one where providers work to correct recommendation inaccuracies from AI-based chatbots. Instead, it should be one of information- and relationship-building where the patient takes a proactive part to work with the healthcare provider and arrive at a final decision. ChatGPT also lacks incentives to improve its performance standards. Human practitioners have moral obligations and liability of malpractice and negligence to provide high standards of care, while this ability is not yet developed in ChatGPT. Since ChatGPT is sensitive to prompt design and has the potential to provide irrelevant information, utilizing ChatGPT properly and reliably requires knowledge and skills to ask the right questions and assess the quality of the information received that not everyone possesses. 16 , 67 , 126 – 128 Although ChatGPT-based information is accessible, it can worsen the existing disparities around health literacy due to this initial knowledge threshold required to use the model. If not done correctly, ChatGPT may miss important information, give generic answers that are not specific to the patient’s situation, or provide misleading or overly worrisome messages. 1 , 129 – 131 As mentioned above, ChatGPT is subject to biases in its training data and memory, both are aspects the patient may not be aware of when obtaining or acting on the medical information provided by ChatGPT. Although ChatGPT has an excellent ability to generate human-like and coherent texts, it cannot think like a medical professional or ask follow-up questions. On the contrary, medical professionals are crucial not only hearing the patients’ experiences, but also offering clarification, actively engaging patients to ensure proper understanding, and providing emotional support. Therefore, if ChatGPT or any AI-based sources were to be implemented in patient education in the future, close monitoring and guidance by medical professionals would be necessary to help patients develop the ability to identify reliable and authoritative sources and critically assess the quality of information online before they can be left on their own. These skills are especially important to develop with the increased democratization of knowledge and autonomy patients have in their own health education. 1 , 85 Gaps in ChatGPT Research Primary studies that evaluate the consistency and quality of ChatGPT’s responses vary in rigorousness. Some studies rate ChatGPT’s responses on a Likert scale while others report the accuracy of ChatGPT responses in percentages. The lack of standardized metrics makes it hard to compare the accuracy of the model across specialties. Mixed results exist in many areas of ChatGPT research. Some showed an internal consistency of 100% and commented highly on its accuracy while others found that even the most updated GPT model (GPT-4) frequently self-contradict, and ChatGPT only has an accuracy of around or less than 50%. 11 , 13 The sensitivity of the model to prompt design, its memory bias, the temperature control, and its constantly evolving nature make it almost impossible to peer review, reproduce, and validate study findings around ChatGPT. Contradicting results across studies can be justified as ChatGPT’s ability to learn and evolve, while it may also be interpreted as a sign of inconsistency in the ChatGPT model. Many other studies compare ChatGPT to Google searches or other AI-based chatbots (Google Bard and BingAI), stating that ChatGPT’s responses are comparable to or more valuable than the ones provided by Google and other AI chatbots. 28 , 62 , 63 , 66 , 106 , 132 – 135 However, this comparison does not that ChatGPT should be used and trusted for medical information. Furthermore, current ChatGPT research lacks a community perspective, using expert-designed or guideline-oriented questions to generate ChatGPT responses and seeking experts’ views on incorporating this in clinical practice. Although healthcare professionals play a crucial role in clinical interactions, they are only part of the story. Patients’ perspectives should also be gathered and incorporated through community outreach efforts. With ChatGPT’s sensitivity to instructions and prompt design, the differences in questions asked between professionals and a layperson without much background in a specific topic area can make a significant difference in the quality of response they get. Future Directions The crucial community perspective is lacking in current ChatGPT research. Only two studies among all that are included in this review sought community’s expertise on this issue. 77 , 136 A crucial next step is to continue this effort, survey patients, and evaluate their experience, trust, and comfortability in utilizing AI-based models for medical information. 137 Although comparison between different versions of GPT models did not reach a consensus, there is early evidence suggesting a superior performance in the paid GPT 4 compared to the publicly accessible ChatGPT (GPT 3.5). This raises concerns from a health equity standpoint, and future research is needed to address this issue. Inadequacies of ChatGPT are not unique to the model and can extend to other AI-based chatbots as well. With Nvidia and Hippocratic AI’s recent collaboration on an empathetic AI agent that outperformed human nurses in offering clinical information and decisions, the speed and scope of such technological developments have been pushed to a new height, making accountability and ethical evaluations of such practices especially pressing and necessary. 138 Although there are some regulations and quality assessment tools in development, such progress is far from catching up with how fast AI-based chatbots has been spread and utilized in daily, academic, and medical settings. 36 Future research should focus on increasing patients’ awareness of the limitations of AI-based chatbots while establishing regulations to ensure ethical and equitable use of such models. Conclusions We evaluated the potential of incorporating ChatGPT in clinical practice and revealed the inconsistency, inaccuracy, and lack of accountability in the ChatGPT model. With significant delays in model training, events of hallucination, and a lack of transparency in its reasoning, ChatGPT is not yet a reliable source of health information. The purpose of patient education and increasing health literacy involves trust- and relationship-building between the patient and the provider, and it exists beyond what ChatGPT can provide. Therefore, we caution against the premature acceptance of ChatGPT as a patient education tool until its logical, legal, and accuracy concerns are addressed. We encourage future efforts to increase transparency of model development and reasoning and incorporate community perspectives in AI research. Data Availability All data produced in the present study are available upon reasonable request to the authors Ethics approval and consent to participate This study is a review of published literature and is exempt from human subjects research review by the Stanford University Institutional Review Board. Availability of data and materials Not applicable (this manuscript does not report data generation or analysis). Conflicts of Interest The authors declare that they have no competing interests. Authors’ contributions Framework and design of the review: R. D. and M, E. C.; literature searches and review: M. C.; drafting the manuscript: M. C. and R. D.; revision of the manuscript: M. C., K. C. P., R. D., and M. E. C.; final approval of the manuscript for publication: M. C., K. C. P., R. D., and M. E. C. Acknowledgement Dr. Reid Dale is supported by Stanford University Department of Cardiothoracic Surgery [grant number 1252717-910-EAGHD]. List of Abbreviations AI Artificial Intelligence ChatGPT Chat Generative Pre-Trained Transformer GPT Generative Pre-Trained Transformers LLM Large Language Model PRISMA Preferred Reporting Items for Systematic Reviews and Meta-Analyses References 1. ↵ Deiana G , Dettori M , Arghittu A , Azara A , Gabutti G , Castiglia P . Artificial Intelligence and Public Health: Evaluating ChatGPT Responses to Vaccination Myths and Misconceptions . Vaccines . 2023 ; 11 ( 7 ): 1217 . doi: 10.3390/vaccines11071217 OpenUrl CrossRef PubMed 2. ↵ Nazir A , Wang Z . A comprehensive survey of ChatGPT: Advancements, applications, prospects, and challenges . Meta-Radiol . 2023 ; 1 ( 2 ): 100022 . doi: 10.1016/j.metrad.2023.100022 OpenUrl CrossRef 3. ↵ Sng GGR , Tung JYM , Lim DYZ , Bee YM . Potential and Pitfalls of ChatGPT and Natural-Language Artificial Intelligence Models for Diabetes Education . Diabetes Care . 2023 ; 46 ( 5 ): e103 – e105 . doi: 10.2337/dc23-0197 OpenUrl CrossRef PubMed 4. ↵ Ray PP . ChatGPT: A comprehensive review on background, applications, key challenges, bias, ethics, limitations and future scope . Internet Things Cyber-Phys Syst . 2023 ; 3 : 121 – 154 . doi: 10.1016/j.iotcps.2023.04.003 OpenUrl CrossRef 5. Frosolini A , Franz L , Benedetti S , et al. Assessing the accuracy of ChatGPT references in head and neck and ENT disciplines . Eur Arch Otorhinolaryngol . 2023 ; 280 ( 11 ): 5129 – 5133 . doi: 10.1007/s00405-023-08205-4 OpenUrl CrossRef PubMed 6. Moshirfar M , Altaf AW , Stoakes IM , Tuttle JJ , Hoopes PC . Artificial Intelligence in Ophthalmology: A Comparative Analysis of GPT-3.5, GPT-4, and Human Expertise in Answering StatPearls Questions . Cureus . Published online June 22, 2023. doi: 10.7759/cureus.40822 OpenUrl CrossRef PubMed 7. ↵ Wang G , Gao K , Liu Q , et al. Potential and Limitations of ChatGPT 3.5 and 4.0 as a Source of COVID-19 Information: Comprehensive Comparative Analysis of Generative and Authoritative Information . J Med Internet Res . 2023 ; 25 : e49771 . doi: 10.2196/49771 OpenUrl CrossRef PubMed 8. ↵ Cascella M , Montomoli J , Bellini V , Bignami E . Evaluating the Feasibility of ChatGPT in Healthcare: An Analysis of Multiple Clinical and Research Scenarios . J Med Syst . 2023 ; 47 ( 1 ): 33 . doi: 10.1007/s10916-023-01925-4 OpenUrl CrossRef PubMed 9. Dossantos J , An J , Javan R . Eyes on AI: ChatGPT’s Transformative Potential Impact on Ophthalmology . Cureus . Published online June 21, 2023 . doi: 10.7759/cureus.40765 OpenUrl CrossRef 10. Karabacak M , Margetis K . Embracing Large Language Models for Medical Applications: Opportunities and Challenges . Cureus. Published online May 21, 2023 . doi: 10.7759/cureus.39305 OpenUrl CrossRef PubMed 11. ↵ Liu J , Wang C , Liu S . Utility of ChatGPT in Clinical Practice . J Med Internet Res . 2023 ; 25 : e48568 . doi: 10.2196/48568 OpenUrl CrossRef PubMed 12. ↵ Seth I , Lim B , Xie Y , et al. Comparing the Efficacy of Large Language Models ChatGPT, BARD, and Bing AI in Providing Information on Rhinoplasty: An Observational Study . Aesthetic Surg J Open Forum . 2023 ; 5 :ojad084. doi: 10.1093/asjof/ojad084 OpenUrl CrossRef 13. ↵ Walker HL , Ghani S , Kuemmerli C , et al. Reliability of Medical Information Provided by ChatGPT: Assessment Against Clinical Guidelines and Patient Information Quality Instrument . J Med Internet Res . 2023 ; 25 : e47479 . doi: 10.2196/47479 OpenUrl CrossRef PubMed 14. ↵ Yeo YH , Samaan JS , Ng WH , et al. Assessing the performance of ChatGPT in answering questions regarding cirrhosis and hepatocellular carcinoma . Clin Mol Hepatol . 2023 ; 29 ( 3 ): 721 – 732 . doi: 10.3350/cmh.2023.0089 OpenUrl CrossRef 15. ↵ Newton W. What is BioGPT and what does it mean for healthcare? Clinical Trials Arena . Published February 9, 2023. Accessed July 19, 2024. https://www.clinicaltrialsarena.com/news/biogpt-healthcare/ 16. ↵ Bjerring JC , Busch J. Artificial Intelligence and Patient-Centered Decision-Making . Philos Technol . 2021 ; 34 ( 2 ): 349 – 371 . doi: 10.1007/s13347-019-00391-6 OpenUrl CrossRef 17. Altamimi I , Altamimi A , Alhumimidi AS , Altamimi A , Temsah MH . Snakebite Advice and Counseling From Artificial Intelligence: An Acute Venomous Snakebite Consultation With ChatGPT . Cureus . Published online June 13, 2023 . doi: 10.7759/cureus.40351 OpenUrl CrossRef PubMed 18. Bernstein IA , Zhang Y (Victor), Govil D , et al. Comparison of Ophthalmologist and Large Language Model Chatbot Responses to Online Patient Eye Care Questions . JAMA Netw Open . 2023 ; 6 ( 8 ): e2330320 . doi: 10.1001/jamanetworkopen.2023.30320 OpenUrl CrossRef 19. Cankurtaran RE , Polat YH , Aydemir NG , Umay E , Yurekli OT . Reliability and Usefulness of ChatGPT for Inflammatory Bowel Diseases: An Analysis for Patients and Healthcare Professionals . Cureus . Published online October 9, 2023 . doi: 10.7759/cureus.46736 OpenUrl CrossRef PubMed 20. Caglar U , Yildiz O , Ozervarli MF , et al. Assessing the Performance of Chat Generative Pretrained Transformer (ChatGPT) in Answering Andrology-Related Questions . Urol Res Pract . 2023 ; 49 ( 6 ): 365 – 369 . doi: 10.5152/tud.2023.23171 OpenUrl CrossRef PubMed 21. Draschl A , Hauer G , Fischerauer SF , et al. Are ChatGPT’s Free-Text Responses on Periprosthetic Joint Infections of the Hip and Knee Reliable and Useful? J Clin Med . 2023 ; 12 ( 20 ): 6655 . doi: 10.3390/jcm12206655 OpenUrl CrossRef 22. ↵ Duran GS , Yurdakurban E , Topsakal KG . The Quality of CLP-Related Information for Patients Provided by ChatGPT . Cleft Palate Craniofacial J. Published online December 21 , 2023 : 10556656231222387 . doi: 10.1177/10556656231222387 OpenUrl CrossRef 23. Endo Y , Sasaki K , Moazzam Z , et al. Quality of ChatGPT Responses to Questions Related To Liver Transplantation . J Gastrointest Surg . 2023 ; 27 ( 8 ): 1716 – 1719 . doi: 10.1007/s11605-023-05714-9 OpenUrl CrossRef PubMed 24. Franco D’Souza R , Amanullah S , Mathew M , Surapaneni KM . Appraising the performance of ChatGPT in psychiatry using 100 clinical case vignettes . Asian J Psychiatry . 2023 ; 89 : 103770 . doi: 10.1016/j.ajp.2023.103770 OpenUrl CrossRef 25. ↵ Gabriel J , Shafik L , Alanbuki A , Larner T . The utility of the ChatGPT artificial intelligence tool for patient education and enquiry in robotic radical prostatectomy . Int Urol Nephrol . 2023 ; 55 ( 11 ): 2717 – 2732 . doi: 10.1007/s11255-023-03729-4 OpenUrl CrossRef PubMed 26. ↵ Hernandez CA , Vazquez Gonzalez AE , Polianovskaia A , et al. The Future of Patient Education: AI-Driven Guide for Type 2 Diabetes . Cureus . Published online November 16, 2023 . doi: 10.7759/cureus.48919 OpenUrl CrossRef PubMed 27. Huang SS , Song Q , Beiting KJ , et al. Fact Check: Assessing the Response of ChatGPT to Alzheimer’s Disease Statements with Varying Degrees of Misinformation . Published online September 7, 2023 . doi: 10.1101/2023.09.04.23294917 OpenUrl Abstract / FREE Full Text 28. ↵ Jeong T , Liu H , Alessandri Bonetti M , Pandya S , Nguyen VT , Egro FM . Revolutionizing patient education: CHATGPT outperforms GOOGLE in answering patient queries on free flap reconstruction . Microsurgery . 2023 ; 43 ( 7 ): 752 – 761 . doi: 10.1002/micr.31106 OpenUrl CrossRef PubMed 29. Köroğlu EY , Fakı S , Beştepe N , et al. A Novel Approach: Evaluating ChatGPT’s Utility for the Management of Thyroid Nodules . Cureus . Published online October 24, 2023 . doi: 10.7759/cureus.47576 OpenUrl CrossRef 30. ↵ Krittanawong C , Virk HUH , Kaplin SL , Wang Z , Sharma S , Jneid H . Assessing the Potential of ChatGPT for Patient Education in Cardiac Catheterization Care . JACC Cardiovasc Interv . 2023 ; 16 ( 12 ): 1551 – 1552 . doi: 10.1016/j.jcin.2023.04.042 OpenUrl CrossRef PubMed 31. ↵ Kuşcu O , Pamuk AE , Sütay Süslü N , Hosal S . Is ChatGPT accurate and reliable in answering questions regarding head and neck cancer? Front Oncol . 2023 ; 13 : 1256459 . doi: 10.3389/fonc.2023.1256459 OpenUrl CrossRef 32. Mika AP , Martin JR , Engstrom SM , Polkowski GG , Wilson JM . Assessing ChatGPT Responses to Common Patient Questions Regarding Total Hip Arthroplasty . J Bone Jt Surg . 2023 ; 105 ( 19 ): 1519 – 1526 . doi: 10.2106/JBJS.23.00209 OpenUrl CrossRef 33. Moazzam Z , Lima HA , Endo Y , Noria S , Needleman B , Pawlik TM . A Paradigm Shift: Online Artificial Intelligence Platforms as an Informational Resource in Bariatric Surgery . Obes Surg . 2023 ; 33 ( 8 ): 2611 – 2614 . doi: 10.1007/s11695-023-06675-3 OpenUrl CrossRef PubMed 34. Pushpanathan K , Lim ZW , Er Yew SM , et al. Popular large language model chatbots’ accuracy, comprehensiveness, and self-awareness in answering ocular symptom queries . iScience . 2023 ; 26 ( 11 ): 108163 . doi: 10.1016/j.isci.2023.108163 OpenUrl CrossRef PubMed 35. ↵ Sallam M . ChatGPT Utility in Healthcare Education, Research, and Practice: Systematic Review on the Promising Perspectives and Valid Concerns . Healthcare . 2023 ; 11 ( 6 ): 887 . doi: 10.3390/healthcare11060887 OpenUrl CrossRef PubMed 36. ↵ Sallam M , Barakat M , Sallam M . Pilot Testing of a Tool to Standardize the Assessment of the Quality of Health Information Generated by Artificial Intelligence-Based Models . Cureus . Published online November 24, 2023 . doi: 10.7759/cureus.49373 OpenUrl CrossRef 37. Sharma S , Pajai S , Prasad R , et al. A Critical Review of ChatGPT as a Potential Substitute for Diabetes Educators . Cureus . Published online May 1, 2023 . doi: 10.7759/cureus.38380 OpenUrl CrossRef 38. Song H , Xia Y , Luo Z , et al. Evaluating the Performance of Different Large Language Models on Health Consultation and Patient Education in Urolithiasis . J Med Syst . 2023 ; 47 ( 1 ): 125 . doi: 10.1007/s10916-023-02021-3 OpenUrl CrossRef 39. Rogasch JMM , Metzger G , Preisler M , et al. ChatGPT: Can You Prepare My Patients for [ 18 F]FDG PET/CT and Explain My Reports? J Nucl Med . 2023 ; 64 ( 12 ): 1876 – 1879 . doi: 10.2967/jnumed.123.266114 OpenUrl Abstract / FREE Full Text 40. Van Bulck L , Moons P . What if your patient switches from Dr. Google to Dr. ChatGPT? A vignette-based survey of the trustworthiness, value, and danger of ChatGPT-generated responses to health questions . Eur J Cardiovasc Nurs . 2024 ; 23 ( 1 ): 95 – 98 . doi: 10.1093/eurjcn/zvad038 OpenUrl CrossRef PubMed 41. ↵ Zalzal HG , Abraham A , Cheng J , Shah RK . Can CHATGPT help patients answer their otolaryngology questions? Laryngoscope Investig Otolaryngol . 2024 ; 9 ( 1 ): e1193 . doi: 10.1002/lio2.1193 OpenUrl CrossRef 42. ↵ Al-Dujaili Z , Omari S , Pillai J , Al Faraj A . Assessing the accuracy and consistency of ChatGPT in clinical pharmacy management: A preliminary analysis with clinical pharmacy experts worldwide . Res Soc Adm Pharm . 2023 ; 19 ( 12 ): 1590 – 1594 . doi: 10.1016/j.sapharm.2023.08.012 OpenUrl CrossRef PubMed 43. Anastasio AT , Mills FB , Karavan MP , Adams SB . Evaluating the Quality and Usability of Artificial Intelligence–Generated Responses to Common Patient Questions in Foot and Ankle Surgery . Foot Ankle Orthop . 2023 ; 8 ( 4 ): 24730114231209919 . doi: 10.1177/24730114231209919 OpenUrl CrossRef PubMed 44. Biswas S , Logan NS , Davies LN , Sheppard AL , Wolffsohn JS . Assessing the utility of ChatGPT as an artificial intelligence based large language model for information to answer questions on myopia . Ophthalmic Physiol Opt . 2023 ; 43 ( 6 ): 1562 – 1570 . doi: 10.1111/opo.13207 OpenUrl CrossRef PubMed 45. Bushuven S , Bentele M , Bentele S , et al. “ ChatGPT, Can You Help Me Save My Child’s Life?” - Diagnostic Accuracy and Supportive Capabilities to Lay Rescuers by ChatGPT in Prehospital Basic Life Support and Paediatric Advanced Life Support Cases – An In-silico Analysis . J Med Syst . 2023 ; 47 ( 1 ): 123 . doi: 10.1007/s10916-023-02019-x OpenUrl CrossRef PubMed 46. ↵ Branum C , Schiavenato M . Can ChatGPT Accurately Answer a PICOT Question? Assessing AI Response to a Clinical Question . Nurse Educ . 2023 ; 48 ( 5 ): 231 – 233 . doi: 10.1097/NNE.0000000000001436 OpenUrl CrossRef PubMed 47. ↵ Garg RK , Urs VL , Agrawal AA , Chaudhary SK , Paliwal V , Kar SK . Exploring the role of ChatGPT in patient care (diagnosis and treatment) and medical research: A systematic review . Health Promot Perspect . 2023 ; 13 ( 3 ): 183 – 191 . doi: 10.34172/hpp.2023.22 OpenUrl CrossRef PubMed 48. Jazi AHD , Mahjoubi M , Shahabi S , et al. Bariatric Evaluation Through AI: a Survey of Expert Opinions Versus ChatGPT-4 (BETA-SEOV) . Obes Surg . 2023 ; 33 ( 12 ): 3971 – 3980 . doi: 10.1007/s11695-023-06903-w OpenUrl CrossRef PubMed 49. McCarthy CJ , Berkowitz S , Ramalingam V , Ahmed M . Evaluation of an Artificial Intelligence Chatbot for Delivery of IR Patient Education Material: A Comparison with Societal Website Content . J Vasc Interv Radiol . 2023 ; 34 ( 10 ): 1760 – 1768 .e32. doi: 10.1016/j.jvir.2023.05.037 OpenUrl CrossRef 50. ↵ Mishra A , Begley SL , Chen A , et al. Exploring the Intersection of Artificial Intelligence and Neurosurgery: Let us be Cautious With ChatGPT . Neurosurgery . 2023 ; 93 ( 6 ): 1366 – 1373 . doi: 10.1227/neu.0000000000002598 OpenUrl CrossRef PubMed 51. Momenaei B , Wakabayashi T , Shahlaee A , et al. Appropriateness and Readability of ChatGPT-4-Generated Responses for Surgical Treatment of Retinal Diseases . Ophthalmol Retina . 2023 ; 7 ( 10 ): 862 – 868 . doi: 10.1016/j.oret.2023.05.022 OpenUrl CrossRef 52. Nastasi AJ , Courtright KR , Halpern SD , Weissman GE . A vignette-based evaluation of ChatGPT’s ability to provide appropriate and equitable medical advice across care contexts . Sci Rep . 2023 ; 13 ( 1 ): 17885 . doi: 10.1038/s41598-023-45223-y OpenUrl CrossRef PubMed 53. Sütcüoğlu BM , Güler M . Appropriateness of premature ovarian insufficiency recommendations provided by ChatGPT . Menopause . 2023 ; 30 ( 10 ): 1033 – 1037 . doi: 10.1097/GME.0000000000002246 OpenUrl CrossRef PubMed 54. Stephens LD , Jacobs JW , Adkins BD , Booth GS . Battle of the (Chat)Bots: Comparing Large Language Models to Practice Guidelines for Transfusion-Associated Graft-Versus-Host Disease Prevention . Transfus Med Rev . 2023 ; 37 ( 3 ): 150753 . doi: 10.1016/j.tmrv.2023.150753 OpenUrl CrossRef PubMed 55. Stroop A , Stroop T , Zawy Alsofy S , et al. Large language models: Are artificial intelligence-based chatbots a reliable source of patient information for spinal surgery? Eur Spine J . Published online October 11, 2023 . doi: 10.1007/s00586-023-07975-z OpenUrl CrossRef 56. Temsah MH , Aljamaan F , Malki KH , et al. ChatGPT and the Future of Digital Health: A Study on Healthcare Workers’ Perceptions and Expectations . Healthcare . 2023 ; 11 ( 13 ): 1812 . doi: 10.3390/healthcare11131812 OpenUrl CrossRef 57. ↵ Wang C , Liu S , Yang H , Guo J , Wu Y , Liu J . Ethical Considerations of Using ChatGPT in Health Care . J Med Internet Res . 2023 ; 25 : e48009 . doi: 10.2196/48009 OpenUrl CrossRef PubMed 58. ↵ Uz C , Umay E . “Dr CHATGPT”: Is it a reliable and useful source for common rheumatic diseases? Int J Rheum Dis . 2023 ; 26 ( 7 ): 1343 – 1349 . doi: 10.1111/1756-185X.14749 OpenUrl CrossRef 59. ↵ Krittanawong C , Rodriguez M , Kaplin S , Tang WHW . Assessing the potential of ChatGPT for patient education in the cardiology clinic . Prog Cardiovasc Dis . 2023 ; 81 : 109 – 110 . doi: 10.1016/j.pcad.2023.10.002 OpenUrl CrossRef PubMed 60. ↵ Dave T , Athaluri SA , Singh S . ChatGPT in medicine: an overview of its applications, advantages, limitations, future prospects, and ethical considerations . Front Artif Intell . 2023 ; 6 : 1169595 . doi: 10.3389/frai.2023.1169595 OpenUrl CrossRef 61. ↵ Mondal H , Mondal S , Podder I . Using ChatGPT for writing articles for patients’ education for dermatological diseases: A pilot study . Indian Dermatol Online J . 2023 ; 14 ( 4 ): 482 . doi: 10.4103/idoj.idoj_72_23 OpenUrl CrossRef PubMed 62. ↵ Ayoub NF , Lee Y , Grimm D , Divi V . Head to Head Comparison of ChatGPT Versus Google Search for Medical Knowledge Acquisition . Otolaryngol Neck Surg . 2024 ; 170 ( 6 ): 1484 – 1491 . doi: 10.1002/ohn.465 OpenUrl CrossRef 63. ↵ Bellinger JR , De La Chapa JS , Kwak MW , Ramos GA , Morrison D , Kesser BW . BPPV Information on Google Versus AI (ChatGPT) . Otolaryngol Neck Surg . 2024 ; 170 ( 6 ): 1504 – 1511 . doi: 10.1002/ohn.506 OpenUrl CrossRef 64. Golan R , Ripps SJ , Reddy R , et al. ChatGPT’s Ability to Assess Quality and Readability of Online Medical Information: Evidence From a Cross-Sectional Study . Cureus . Published online July 20, 2023 . doi: 10.7759/cureus.42214 OpenUrl CrossRef PubMed 65. Haidar O , Jaques A , McCaughran PW , Metcalfe MJ . AI-Generated Information for Vascular Patients: Assessing the Standard of Procedure-Specific Information Provided by the ChatGPT AI-Language Model . Cureus . Published online November 30, 2023 . doi: 10.7759/cureus.49764 OpenUrl CrossRef 66. ↵ Hristidis V , Ruggiano N , Brown EL , Ganta SRR , Stewart S . ChatGPT vs Google for Queries Related to Dementia and Other Cognitive Decline: Comparison of Results . J Med Internet Res . 2023 ; 25 : e48966 . doi: 10.2196/48966 OpenUrl CrossRef PubMed 67. ↵ Spallek S , Birrell L , Kershaw S , Devine EK , Thornton L . Can we use ChatGPT for Mental Health and Substance Use Education? Examining Its Quality and Potential Harms . JMIR Med Educ . 2023 ; 9 : e51243 . doi: 10.2196/51243 OpenUrl CrossRef 68. Ulusoy I , Yılmaz M , Kıvrak A . How Efficient Is ChatGPT in Accessing Accurate and Quality Health-Related Information? Cureus . Published online October 7, 2023 . doi: 10.7759/cureus.46662 OpenUrl CrossRef 69. ↵ Yurdakurban E , Topsakal KG , Duran GS . A comparative analysis of AI-based chatbots: Assessing data quality in orthognathic surgery related patient information . J Stomatol Oral Maxillofac Surg . 2024 ; 125 ( 5 ): 101757 . doi: 10.1016/j.jormas.2023.101757 OpenUrl CrossRef PubMed 70. ↵ Whitney L. ChatGPT is no longer as clueless about recent events | ZDNET . ZDNET . Published November 7, 2023. Accessed July 19, 2024. https://www.zdnet.com/article/chatgpt-is-no-longer-as-clueless-about-recent-events/ 71. ↵ Grünebaum A , Chervenak J , Pollet SL , Katz A , Chervenak FA . The exciting potential for ChatGPT in obstetrics and gynecology . Am J Obstet Gynecol . 2023 ; 228 ( 6 ): 696 – 705 . doi: 10.1016/j.ajog.2023.03.009 OpenUrl CrossRef PubMed 72. ↵ Peeperkorn M , Kouwenhoven T , Brown D , Jordanous A . Is Temperature the Creativity Parameter of Large Language Models? Published online May 1 , 2024 . Accessed July 19, 2024. http://arxiv.org/abs/2405.00492 73. ↵ Watkins R . Guidance for researchers and peer-reviewers on the ethical use of Large Language Models (LLMs) in scientific research workflows . AI Ethics . Published online May 16, 2023 : s 43681-023-00294-00295. doi: 10.1007/s43681-023-00294-5 OpenUrl CrossRef 74. ↵ Jeyaraman M , Balaji S , Jeyaraman N , Yadav S . Unraveling the Ethical Enigma: Artificial Intelligence in Healthcare . Cureus . Published online August 10, 2023 . doi: 10.7759/cureus.43262 OpenUrl CrossRef 75. ↵ Victor G , Bélisle-Pipon JC , Ravitsky V. Generative AI, Specific Moral Values: A Closer Look at ChatGPT’s New Ethical Implications for Medical AI . Am J Bioeth . 2023 ; 23 ( 10 ): 65 – 68 . doi: 10.1080/15265161.2023.2250311 OpenUrl CrossRef 76. ↵ Tran TH. OpenAI’s ChatGPT Went Completely Off the Rails for Hours . The Daily Beast . https://www.thedailybeast.com/openais-chatgpt-went-completely-off-the-rails-for-hours . Published February 21, 2024. Accessed February 21, 2024 . 77. ↵ Nov O , Singh N , Mann D . Putting ChatGPT’s Medical Advice to the (Turing) Test: Survey Study . JMIR Med Educ . 2023 ; 9 : e46939 . doi: 10.2196/46939 OpenUrl CrossRef 78. ↵ Hung YC , Chaker SC , Sigel M , Saad M , Slater ED . Comparison of Patient Education Materials Generated by Chat Generative Pre-Trained Transformer Versus Experts: An Innovative Way to Increase Readability of Patient Education Materials . Ann Plast Surg . 2023 ; 91 ( 4 ): 409 – 412 . doi: 10.1097/SAP.0000000000003634 OpenUrl CrossRef PubMed 79. ↵ Athavale A , Baier J , Ross E , Fukaya E . The potential of chatbots in chronic venous disease patient management . JVS-Vasc Insights . 2023 ; 1 : 100019 . doi: 10.1016/j.jvsvi.2023.100019 OpenUrl CrossRef 80. Klang E , Sourosh A , Nadkarni GN , Sharif K , Lahat A . Evaluating the role of ChatGPT in gastroenterology: a comprehensive systematic review of applications, benefits, and limitations . Ther Adv Gastroenterol . 2023 ; 16 :17562848231218618. doi: 10.1177/17562848231218618 OpenUrl CrossRef PubMed 81. Minssen T , Vayena E , Cohen IG . The Challenges for Regulating Medical Use of ChatGPT and Other Large Language Models . JAMA . 2023 ; 330 ( 4 ): 315 . doi: 10.1001/jama.2023.9651 OpenUrl CrossRef PubMed 82. Wilhelm TI , Roos J , Kaczmarczyk R . Large Language Models for Therapy Recommendations Across 3 Clinical Specialties: Comparative Study . J Med Internet Res . 2023 ; 25 : e49324 . doi: 10.2196/49324 OpenUrl CrossRef 83. ↵ Yun JY , Kim DJ , Lee N , Kim EK . A comprehensive evaluation of ChatGPT consultation quality for augmentation mammoplasty: A comparative analysis between plastic surgeons and laypersons . Int J Med Inf . 2023 ; 179 : 105219 . doi: 10.1016/j.ijmedinf.2023.105219 OpenUrl CrossRef PubMed 84. ↵ Nasr M , Carlini N , Hayase J , et al. Scalable Extraction of Training Data from (Production) Language Models . Published online 2023. doi: 10.48550/ARXIV.2311.17035 OpenUrl CrossRef 85. ↵ Johnson D , Goodman R , Patrinely J , et al. Assessing the Accuracy and Reliability of AI-Generated Medical Responses: An Evaluation of the Chat-GPT Model . Published online February 28, 2023 . doi: 10.21203/rs.3.rs-2566942/v1 OpenUrl CrossRef PubMed 86. ↵ Salas A , Rivero-Calle I , Martinón-Torres F . Chatting with ChatGPT to learn about safety of COVID-19 vaccines – A perspective . Hum Vaccines Immunother . 19 ( 2 ): 2235200 . doi: 10.1080/21645515.2023.2235200 OpenUrl CrossRef 87. ↵ Meyrowitsch DW , Jensen AK , Sørensen JB , Varga TV . AI chatbots and (mis)information in public health: impact on vulnerable communities . Front Public Health . 2023 ; 11 :1226776. doi: 10.3389/fpubh.2023.1226776 OpenUrl CrossRef 88. ↵ Zhuo TY , Huang Y , Chen C , Xing Z . Red teaming ChatGPT via Jailbreaking: Bias, Robustness, Reliability and Toxicity . Published online May 29, 2023 . Accessed July 19, 2024. http://arxiv.org/abs/2301.12867 89. ↵ Jang ME , Lukasiewicz T. Consistency Analysis of ChatGPT . Published online 2023. doi: 10.48550/ARXIV.2303.06273 OpenUrl CrossRef 90. ↵ Ferro Desideri L , Roth J , Zinkernagel M , Anguita R . Application and accuracy of artificial intelligence-derived large language models in patients with age related macular degeneration . Int J Retina Vitr . 2023 ; 9 ( 1 ): 71 . doi: 10.1186/s40942-023-00511-7 OpenUrl CrossRef 91. ↵ Goodman RS , Patrinely JR , Stone CA , et al. Accuracy and Reliability of Chatbot Responses to Physician Questions . JAMA Netw Open . 2023 ; 6 ( 10 ): e2336483 . doi: 10.1001/jamanetworkopen.2023.36483 OpenUrl CrossRef 92. ↵ Tanaka OM , Gasparello GG , Hartmann GC , Casagrande FA , Pithon MM . Assessing the reliability of ChatGPT: a content analysis of self-generated and self-answered questions on clear aligners , TADs and digital imaging. Dent Press J Orthod . 2023 ; 28 ( 5 ): e2323183 . doi: 10.1590/2177-6709.28.5.e2323183.oar OpenUrl CrossRef 93. ↵ Sharun K , Banu SA , Pawde AM , et al. ChatGPT and artificial hallucinations in stem cell research: assessing the accuracy of generated references – a preliminary study . Ann Med Surg . 2023 ; 85 ( 10 ): 5275 – 5278 . doi: 10.1097/MS9.0000000000001228 OpenUrl CrossRef 94. ↵ Sridi C , Brigui S . The use of ChatGPT in occupational medicine: opportunities and threats . Ann Occup Environ Med . 2023 ; 35 ( 1 ): e42 . doi: 10.35371/aoem.2023.35.e42 OpenUrl CrossRef 95. ↵ Potapenko I , Boberg Ans LC , Stormly Hansen M , Klefter ON , Van Dijk EHC , Subhi Y . Artificial intelligence based chatbot patient information on common retinal diseases using CHATGPT . Acta Ophthalmol (Copenh ) . 2023 ; 101 ( 7 ): 829 – 831 . doi: 10.1111/aos.15661 OpenUrl CrossRef 96. Cao JJ , Kwon DH , Ghaziani TT , et al. Accuracy of Information Provided by ChatGPT Regarding Liver Cancer Surveillance and Diagnosis . Am J Roentgenol . 2023 ; 221 ( 4 ): 556 – 559 . doi: 10.2214/AJR.23.29493 OpenUrl CrossRef PubMed 97. Sun H , Zhang K , Lan W , et al. An AI Dietitian for Type 2 Diabetes Mellitus Management Based on Large Language and Image Recognition Models: Preclinical Concept Validation Study . J Med Internet Res . 2023 ; 25 : e51300 . doi: 10.2196/51300 OpenUrl CrossRef PubMed 98. ↵ Shao C ye, Li H , Liu X long, et al. Appropriateness and Comprehensiveness of Using ChatGPT for Perioperative Patient Education in Thoracic Surgery in Different Language Contexts: Survey Study . Interact J Med Res . 2023 ; 12 : e46900 . doi: 10.2196/46900 OpenUrl CrossRef 99. ↵ Whiles BB , Bird VG , Canales BK , DiBianco JM , Terry RS . Caution! AI Bot Has Entered the Patient Chat: ChatGPT Has Limitations in Providing Accurate Urologic Healthcare Advice . Urology . 2023 ; 180 : 278 – 284 . doi: 10.1016/j.urology.2023.07.010 OpenUrl CrossRef 100. ↵ Wang G , Liu Q , Chen G , et al. AI’s deep dive into complex pediatric inguinal hernia issues: a challenge to traditional guidelines? Hernia . 2023 ; 27 ( 6 ): 1587 – 1599 . doi: 10.1007/s10029-023-02900-1 OpenUrl CrossRef PubMed 101. ↵ Acerbi A , Stubbersfield JM . Large language models show human-like content biases in transmission chain experiments . Proc Natl Acad Sci . 2023 ; 120 ( 44 ): e2313790120 . doi: 10.1073/pnas.2313790120 OpenUrl CrossRef PubMed 102. ↵ Koranteng E , Rao A , Flores E , et al. Empathy and Equity: Key Considerations for Large Language Model Adoption in Health Care . JMIR Med Educ . 2023 ; 9 : e51199 . doi: 10.2196/51199 OpenUrl CrossRef 103. ↵ Haze T , Kawano R , Takase H , Suzuki S , Hirawa N , Tamura K . Influence on the accuracy in ChatGPT: Differences in the amount of information per medical field . Int J Med Inf . 2023 ; 180 : 105283 . doi: 10.1016/j.ijmedinf.2023.105283 OpenUrl CrossRef PubMed 104. ↵ Bhattacharyya M , Miller VM , Bhattacharyya D , Miller LE . High Rates of Fabricated and Inaccurate References in ChatGPT-Generated Medical Content . Cureus . 2023 ; 15 ( 5 ): e39238 . doi: 10.7759/cureus.39238 OpenUrl CrossRef 105. ↵ McGowan A , Gui Y , Dobbs M , et al. ChatGPT and Bard exhibit spontaneous citation fabrication during psychiatry literature search . Psychiatry Res . 2023 ; 326 : 115334 . doi: 10.1016/j.psychres.2023.115334 OpenUrl CrossRef PubMed 106. ↵ Alessandri-Bonetti M , Liu HY , Palmesano M , Nguyen VT , Egro FM . Online patient education in body contouring: A comparison between Google and ChatGPT . J Plast Reconstr Aesthet Surg . 2023 ; 87 : 390 – 402 . doi: 10.1016/j.bjps.2023.10.091 OpenUrl CrossRef PubMed 107. Chervenak J , Lieman H , Blanco-Breindel M , Jindal S . The promise and peril of using a large language model to obtain clinical information: ChatGPT performs strongly as a fertility counseling tool with limitations . Fertil Steril . 2023 ; 120 ( 3 ): 575 – 583 . doi: 10.1016/j.fertnstert.2023.05.151 OpenUrl CrossRef 108. Copeland-Halperin LR , O’Brien L , Copeland M . Evaluation of Artificial Intelligence– generated Responses to Common Plastic Surgery Questions . Plast Reconstr Surg - Glob Open . 2023 ; 11 ( 8 ): e5226 . doi: 10.1097/GOX.0000000000005226 OpenUrl CrossRef 109. Crook BS , Park CN , Hurley ET , Richard MJ , Pidgeon TS . Evaluation of Online Artificial Intelligence-Generated Information on Common Hand Procedures . J Hand Surg . 2023 ; 48 ( 11 ): 1122 – 1127 . doi: 10.1016/j.jhsa.2023.08.003 OpenUrl CrossRef 110. Randhawa J , Khan A . A Conversation With ChatGPT About the Usage of Lithium in Pregnancy for Bipolar Disorder . Cureus . Published online October 5, 2023 . doi: 10.7759/cureus.46548 OpenUrl CrossRef 111. Sevgi UT , Erol G , Doğruel Y , Sönmez OF , Tubbs RS , Güngor A . The role of an open artificial intelligence platform in modern neurosurgical education: a preliminary study . Neurosurg Rev . 2023 ; 46 ( 1 ): 86 . doi: 10.1007/s10143-023-01998-2 OpenUrl CrossRef 112. Suppadungsuk S , Thongprayoon C , Krisanapan P , et al. Examining the Validity of ChatGPT in Identifying Relevant Nephrology Literature: Findings and Implications . J Clin Med . 2023 ; 12 ( 17 ): 5550 . doi: 10.3390/jcm12175550 OpenUrl CrossRef PubMed 113. Vaira LA , Lechien JR , Abbate V , et al. Accuracy of ChatGPT Generated Information on Head and Neck and Oromaxillofacial Surgery: A Multicenter Collaborative Analysis . Otolaryngol Neck Surg . 2024 ; 170 ( 6 ): 1492 – 1503 . doi: 10.1002/ohn.489 OpenUrl CrossRef 114. ↵ Warren E , Hurley ET , Park CN , et al. Evaluation of information from artificial intelligence on rotator cuff repair surgery . JSES Int . 2024 ; 8 ( 1 ): 53 – 57 . doi: 10.1016/j.jseint.2023.09.009 OpenUrl CrossRef PubMed 115. ↵ Abbasi J , Hswen Y . How AI Assistants Could Help Answer Patients’ Messages—and Potentially Improve Their Outcomes . JAMA . 2024 ; 331 ( 2 ): 95 . doi: 10.1001/jama.2023.22555 OpenUrl CrossRef PubMed 116. Alten A , Gündeş E , Tuncer E , Kozanoğlu E , Akalın BE , Emekli U . Integrating artificial intelligence in orthognathic surgery: A case study of ChatGPT’s role in enhancing physician-patient consultations for dentofacial deformities . J Plast Reconstr Aesthet Surg . 2023 ; 87 : 405 – 407 . doi: 10.1016/j.bjps.2023.10.097 OpenUrl CrossRef PubMed 117. Holmes J , Liu Z , Zhang L , et al. Evaluating large language models on a highly-specialized topic, radiation oncology physics . Front Oncol . 2023 ; 13 : 1219326 . doi: 10.3389/fonc.2023.1219326 OpenUrl CrossRef PubMed 118. Jin Y , Liu H , Zhao B , Pan W . ChatGPT and mycosis– a new weapon in the knowledge battlefield . BMC Infect Dis . 2023 ; 23 ( 1 ): 731 . doi: 10.1186/s12879-023-08724-9 OpenUrl CrossRef PubMed 119. Marcantonio TL , Nielsen KE , Haikalis M , et al. Hey ChatGPT, Let’s Talk About Sexual Consent . J Sex Res. Published online September 14 , 2023 : 1 – 12 . doi: 10.1080/00224499.2023.2254772 OpenUrl CrossRef 120. Wang A , Kim E , Oleru O , Seyidova N , Taub PJ . Artificial Intelligence in Plastic Surgery: ChatGPT as a Tool to Address Disparities in Health Literacy . Plast Reconstr Surg . 2024 ; 153 ( 6 ): 1232e – 1234e . doi: 10.1097/PRS.0000000000011202 OpenUrl CrossRef PubMed 121. Wójcik S , Rulkiewicz A , Pruszczyk P , Lisik W , Poboży M , Domienik-Karłowicz J. Beyond ChatGPT: What does GPT-4 add to healthcare? The dawn of a new era . Cardiol J . Published online October 12, 2023 :VM/OJS/J/97515. doi: 10.5603/cj.97515 OpenUrl CrossRef PubMed 122. ↵ Samaan JS , Yeo YH , Rajeev N , et al. Assessing the Accuracy of Responses by the Language Model ChatGPT to Questions Regarding Bariatric Surgery . Obes Surg . 2023 ; 33 ( 6 ): 1790 – 1796 . doi: 10.1007/s11695-023-06603-5 OpenUrl CrossRef 123. ↵ Health Communication - Healthy People 2030 | health.gov. Accessed July 19, 2024. https://health.gov/healthypeople/objectives-and-data/browse-objectives/health-communication 124. ↵ Health Literacy in Healthy People 2030 - Healthy People 2030 | health.gov. Accessed July 19, 2024. https://health.gov/healthypeople/priority-areas/health-literacy-healthy-people-2030 125. Wittink H , Oosterhaven J. Patient education and health literacy . Musculoskelet Sci Pract . 2018 ; 38 : 120 – 127 . doi: 10.1016/j.msksp.2018.06.004 OpenUrl CrossRef PubMed 126. ↵ Schulz PJ , Nakamoto K . Patient behavior and the benefits of artificial intelligence: The perils of “dangerous” literacy and illusory patient empowerment . Patient Educ Couns . 2013 ; 92 ( 2 ): 223 – 228 . doi: 10.1016/j.pec.2013.05.002 OpenUrl CrossRef PubMed 127. Bilika P , Stefanouli V , Strimpakos N , Kapreli EV . Clinical reasoning using ChatGPT: Is it beyond credibility for physiotherapists use? Physiother Theory Pract . Published online December 11, 2023 : 1 – 20 . doi: 10.1080/09593985.2023.2291656 OpenUrl CrossRef 128. ↵ Lin Z . Why and how to embrace AI such as ChatGPT in your academic life . R Soc Open Sci . 2023 ; 10 ( 8 ): 230658 . doi: 10.1098/rsos.230658 OpenUrl CrossRef PubMed 129. ↵ Hillmann HAK , Angelini E , Karfoul N , Feickert S , Mueller-Leisse J , Duncker D . Accuracy and comprehensibility of chat-based artificial intelligence for patient information on atrial fibrillation and cardiac implantable electronic devices . Europace . 2023 ; 26 ( 1 ): euad369 . doi: 10.1093/europace/euad369 OpenUrl CrossRef PubMed 130. Kuang YR , Zou MX , Niu HQ , Zheng BY , Zhang TL , Zheng BW . ChatGPT encounters multiple opportunities and challenges in neurosurgery . Int J Surg . 2023 ; 109 ( 10 ): 2886 – 2891 . doi: 10.1097/JS9.0000000000000571 OpenUrl CrossRef PubMed 131. ↵ Lautrup AD , Hyrup T , Schneider-Kamp A , Dahl M , Lindholt JS , Schneider-Kamp P . Heart-to-heart with ChatGPT: the impact of patients consulting AI for cardiovascular health advice . Open Heart . 2023 ; 10 ( 2 ): e002455 . doi: 10.1136/openhrt-2023-002455 OpenUrl CrossRef PubMed 132. ↵ Ayoub NF , Lee YJ , Grimm D , Balakrishnan K . Comparison Between ChatGPT and Google Search as Sources of Postoperative Patient Instructions . JAMA Otolaryngol Neck Surg . 2023 ; 149 ( 6 ): 556 . doi: 10.1001/jamaoto.2023.0704 OpenUrl CrossRef 133. Lim ZW , Pushpanathan K , Yew SME , et al. Benchmarking large language models’ performances for myopia care: a comparative analysis of ChatGPT-3.5, ChatGPT-4.0, and Google Bard . eBioMedicine . 2023 ; 95 : 104770 . doi: 10.1016/j.ebiom.2023.104770 OpenUrl CrossRef PubMed 134. Mu X , Lim B , Seth I , et al. Comparison of large language models in management advice for melanoma: Google’s AI BARD, BingAI and ChatGPT . Skin Health Dis . 2024 ; 4 ( 1 ): e313 . doi: 10.1002/ski2.313 OpenUrl CrossRef 135. ↵ Rahsepar AA , Tavakoli N , Kim GHJ , Hassani C , Abtin F , Bedayat A . How AI Responds to Common Lung Cancer Questions: ChatGPT versus Google Bard . Radiology . 2023 ; 307 ( 5 ): e230922 . doi: 10.1148/radiol.230922 OpenUrl CrossRef PubMed 136. ↵ Shahsavar Y , Choudhury A . User Intentions to Use ChatGPT for Self-Diagnosis and Health-Related Purposes: Cross-sectional Survey Study . JMIR Hum Factors . 2023 ; 10 : e47564 . doi: 10.2196/47564 OpenUrl CrossRef 137. ↵ Pillai M , Griffin AC , Kronk CA , McCall T . Toward Community-Based Natural Language Processing (CBNLP): Cocreating With Communities . J Med Internet Res . 2023 ; 25 : e48498 . doi: 10.2196/48498 OpenUrl CrossRef PubMed 138. ↵ Dumas B. Nvidia announces AI-powered health care “agents” that outperform nurses — and cost $9 an hour . FOXBusiness . Published March 21, 2024. Accessed July 19, 2024. https://www.foxbusiness.com/technology/nvidia-announces-ai-powered-health-care-agents-outperform-nurses-cost-9-hour View the discussion thread. Back to top Previous Next Posted May 21, 2025. Download PDF Data/Code Email Thank you for your interest in spreading the word about medRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following The Scope and Limitations of Extant Research into ChatGPT as a Tool for Patient Education: Systematic Review Message Subject (Your Name) has forwarded a page to you from medRxiv Message Body (Your Name) thought you would like to see this page from the medRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share The Scope and Limitations of Extant Research into ChatGPT as a Tool for Patient Education: Systematic Review Reid Dale , Maggie Cheng , Katharine Casselman Pines , Maria Elizabeth Currie medRxiv 2025.05.20.25328009; doi: https://doi.org/10.1101/2025.05.20.25328009 Share This Article: Copy Citation Tools The Scope and Limitations of Extant Research into ChatGPT as a Tool for Patient Education: Systematic Review Reid Dale , Maggie Cheng , Katharine Casselman Pines , Maria Elizabeth Currie medRxiv 2025.05.20.25328009; doi: https://doi.org/10.1101/2025.05.20.25328009 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Health Informatics Subject Areas All Articles Addiction Medicine (568) Allergy and Immunology (863) Anesthesia (299) Cardiovascular Medicine (4425) Dentistry and Oral Medicine (443) Dermatology (382) Emergency Medicine (607) Endocrinology (including Diabetes Mellitus and Metabolic Disease) (1507) Epidemiology (15221) Forensic Medicine (30) Gastroenterology (1123) Genetic and Genomic Medicine (6588) Geriatric Medicine (667) Health Economics (997) Health Informatics (4524) Health Policy (1368) Health Systems and Quality Improvement (1612) Hematology (540) HIV/AIDS (1264) Infectious Diseases (except HIV/AIDS) (15910) Intensive Care and Critical Care Medicine (1103) Medical Education (623) Medical Ethics (145) Nephrology (667) Neurology (6588) Nursing (346) Nutrition (998) Obstetrics and Gynecology (1143) Occupational and Environmental Health (956) Oncology (3331) Ophthalmology (970) Orthopedics (369) Otolaryngology (420) Pain Medicine (435) Palliative Medicine (129) Pathology (663) Pediatrics (1690) Pharmacology and Therapeutics (691) Primary Care Research (710) Psychiatry and Clinical Psychology (5440) Public and Global Health (9219) Radiology and Imaging (2195) Rehabilitation Medicine and Physical Therapy (1369) Respiratory Medicine (1196) Rheumatology (593) Sexual and Reproductive Health (710) Sports Medicine (529) Surgery (710) Toxicology (99) Transplantation (289) Urology (265) (function(){function c(){var b=a.contentDocument||a.contentWindow.document;if(b){var d=b.createElement('script');d.innerHTML="window.__CF$cv$params={r:'9ffb00d7b9b3c13d',t:'MTc3OTQ0NDMzNQ=='};var a=document.createElement('script');a.src='/cdn-cgi/challenge-platform/scripts/jsd/main.js';document.getElementsByTagName('head')[0].appendChild(a);";b.getElementsByTagName('head')[0].appendChild(d)}}if(document.body){var a=document.createElement('iframe');a.height=1;a.width=1;a.style.position='absolute';a.style.top=0;a.style.left=0;a.style.border='none';a.style.visibility='hidden';document.body.appendChild(a);if('loading'!==document.readyState)c();else if(window.addEventListener)document.addEventListener('DOMContentLoaded',c);else{var e=document.onreadystatechange||function(){};document.onreadystatechange=function(b){e(b);'loading'!==document.readyState&&(document.onreadystatechange=e,c())}}}})();

Text is read by the "Ask this paper" AI Q&A widget below. Extraction quality varies by source — PMC NXML preserves structure cleanly, OA-HTML may include some navigation residue, and OA-PDF can have broken hyphenation. The publisher copy (via DOI) is the canonical version.

My notes (saved in your browser only)

Ask this paper AI returns verbatim quotes from the full text · source: preprint-html

Answers must be backed by verbatim quotes from this paper's full text. Hallucinated quotes are dropped automatically; if no verbatim passage answers the question, we say so. How this works

Citation neighborhood (no data yet)

We don't have any in-corpus citations linked to this paper yet. This is a recent paper (2025) — citers typically take a year or two to land, and the OpenAlex reference graph may still be filling in.

Source provenance

europepmc
last seen: 2026-05-20T01:45:00.602351+00:00