Full text
47,733 characters
· extracted from
preprint-html
· click to expand
PH-LLM: Public Health Large Language Models for Infoveillance | medRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-P4HH5NV'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search PH-LLM: Public Health Large Language Models for Infoveillance View ORCID Profile Xinyu Zhou , Jiaqi Zhou , Chiyu Wang , Qianqian Xie , Kaize Ding , Chengsheng Mao , Yuntian Liu , Zhiyuan Cao , Huangrui Chu , View ORCID Profile Xi Chen , Hua Xu , Heidi J. Larson , Yuan Luo doi: https://doi.org/10.1101/2025.02.08.25321587 Xinyu Zhou 1 Division of Biostatistics and Informatics, Department of Preventive Medicine, Northwestern University , Chicago, IL 60611, USA 2 Health Science Integrated PhD Program, Feinberg School of Medicine, Northwestern University , Chicago, IL, 60611, USA MS Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Xinyu Zhou Jiaqi Zhou 1 Division of Biostatistics and Informatics, Department of Preventive Medicine, Northwestern University , Chicago, IL 60611, USA 2 Health Science Integrated PhD Program, Feinberg School of Medicine, Northwestern University , Chicago, IL, 60611, USA MS Find this author on Google Scholar Find this author on PubMed Search for this author on this site Chiyu Wang 3 Department of Computer Science, Yale University , New Haven, CT 06511, USA MS Find this author on Google Scholar Find this author on PubMed Search for this author on this site Qianqian Xie 4 Department of Biomedical Informatics & Data Science, Yale School of Medicine , CT 06510, USA PhD Find this author on Google Scholar Find this author on PubMed Search for this author on this site Kaize Ding 5 Department of Statistics and Data Science, Northwestern University , Evanston, IL 60208, USA PhD Find this author on Google Scholar Find this author on PubMed Search for this author on this site Chengsheng Mao 1 Division of Biostatistics and Informatics, Department of Preventive Medicine, Northwestern University , Chicago, IL 60611, USA PhD Find this author on Google Scholar Find this author on PubMed Search for this author on this site Yuntian Liu 4 Department of Biomedical Informatics & Data Science, Yale School of Medicine , CT 06510, USA MS Find this author on Google Scholar Find this author on PubMed Search for this author on this site Zhiyuan Cao 4 Department of Biomedical Informatics & Data Science, Yale School of Medicine , CT 06510, USA MS Find this author on Google Scholar Find this author on PubMed Search for this author on this site Huangrui Chu 6 Department of Biostatistics, Yale School of Public Health , New Haven, CT 06510, USA MS Find this author on Google Scholar Find this author on PubMed Search for this author on this site Xi Chen 7 Department of Health Policy and Management, Yale School of Public Health , New Haven, CT 06510, USA 8 Department of Economics, Yale University , New Haven, CT 06511, USA PhD Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Xi Chen Hua Xu 4 Department of Biomedical Informatics & Data Science, Yale School of Medicine , CT 06510, USA PhD Find this author on Google Scholar Find this author on PubMed Search for this author on this site Heidi J. Larson 9 Department of Infectious Disease Dynamics, London School of Hygiene and Tropical Medicine , London W1E 7HT, UK 10 Institute for Health Metrics and Evaluation, University of Washington , Seattle, WA 98195, USA PhD Find this author on Google Scholar Find this author on PubMed Search for this author on this site Yuan Luo 1 Division of Biostatistics and Informatics, Department of Preventive Medicine, Northwestern University , Chicago, IL 60611, USA 11 Center for Collaborative AI in Healthcare, Institute for AI in Medicine, Feinberg School of Medicine, Northwestern University , Chicago, IL 60611, USA PhD Find this author on Google Scholar Find this author on PubMed Search for this author on this site For correspondence: yuan.luo{at}northwestern.edu Abstract Full Text Info/History Metrics Supplementary material Data/Code Preview PDF Summary Background The effectiveness of public health intervention, such as vaccination and social distancing, relies on public support and adherence. Social media has emerged as a critical platform for understanding and fostering public engagement with health interventions. However, the lack of real-time surveillance on public health issues leveraging social media data, particularly during public health emergencies, leads to delayed responses and suboptimal policy adjustments. Methods To address this gap, we developed PH-LLM (Public Health Large Language Models for Infoveillance)—a novel suite of large language models (LLMs) specifically designed for real-time public health monitoring. We curated a multilingual training corpus comprising 593,100 instruction-output pairs from 36 datasets, covering 96 public health infoveillance tasks and 6 question-answering datasets based on social media data. PH-LLM was trained using quantized low-rank adapters (QLoRA) and LoRA plus, leveraging Qwen 2.5, which supports 29 languages. The PH-LLM suite includes models of six different sizes: 0.5B, 1.5B, 3B, 7B, 14B, and 32B. To evaluate PH-LLM, we constructed a benchmark comprising 19 English and 20 multilingual public health tasks using 10 social media datasets (totaling 52,158 unseen instruction-output pairs). We compared PH-LLM’s performance against leading open-source models, including Llama-3.1-70B-Instruct, Mistral-Large-Instruct-2407, and Qwen2.5-72B-Instruct, as well as proprietary models such as GPT-4o. Findings Across 19 English and 20 multilingual evaluation tasks, PH-LLM consistently outperformed baseline models of similar and larger sizes, including instruction-tuned versions of Qwen2.5, Llama3.1/3.2, Mistral, and bloomz, with PH-LLM-32B achieving the state-of-the-art results. Notably, PH-LLM-14B and PH-LLM-32B surpassed Qwen2.5-72B-Instruct, Llama-3.1-70B-Instruct, Mistral-Large-Instruct-2407, and GPT-4o in both English tasks (>=56.0% vs. =59.6% vs. <= 59.1%). The only exception was PH-LLM-7B, with slightly suboptimal average performance (48.7%) in English tasks compared to Qwen2.5-7B-Instruct (50.7%), although it outperformed GPT-4o mini (46.9%), Mistral-Small-Instruct-2409 (45.8%), Llama-3.1-8B-Instruct (45.4%), and bloomz-7b1-mt (27.9%). Interpretation PH-LLM represents a significant advancement in real-time public health infoveillance, offering state-of-the-art multilingual capabilities and cost-effective solutions for monitoring public sentiment on health issues. By equipping global, national, and local public health agencies with timely insights from social media data, PH-LLM has the potential to enhance rapid response strategies, improve policy-making, and strengthen public health communication during crises and beyond. Funding This study is supported in part by NIH grants R01LM013337 (YL). Introduction The effectiveness of public health interventions, such as social distancing, COVID-19 testing, and vaccination, hinges on collective support, participation, and adherence, in both physically spaces and virtual platforms. With recent advances in machine learning, infoveillance—the continuous analysis of online text information 1 —has emerged as a supplement to traditional public health surveillance approaches, offering early insights into public responses to interventions. Infoveillance has also been employed to mitigate the infodemic–an overwhelming surge of information and misinformation that may lead to deleterious public health consequences during pandemics, in addition to ensuring public adherence and informing health policy decisions. 1 – 5 A growing number of public health researchers and authorities are leveraging social media data to explore vaccine attitudes, mental health issues, adherence to non-pharmaceutical interventions (NPIs), the spread of misinformation, and beyond, with machine learning models such as random forest and naïve bayes. 2 , 6 , 7 Despite these efforts, real-time infoveillance on social media platforms remains limited, especially when tracking rapidly evolving public health emergencies like COVID-19 6 , 7 . Without timely and scalable infoveillance methods, there may potentially be delays in policy refinement and missed opportunities for prompt public health interventions. 7 Large Language Models (LLMs) hold promising potentials for infoveillance 8 – 13 . They can perform infoveillance tasks without necessitating the extensive resources, time, and task-specific annotated datasets typically required for training conventional machine learning models for large-scale infoveillance. Moreover, their human-like interactions make them more accessible to public health experts than many other machine learning tools. However, proprietary LLMs such as ChatGPT are associated with significant cost and may lead to data leakage. On the other hand, open-source LLMs targeted general tasks are not optimized for public health inforveillance. There’s a need for developing LLMs tailored for public health infoveillance, which could significantly reduce costs while delivering state-of-the-art performance. In this study, we introduce PH-LLM (Public Health Large Language Models for infoveillance), which is a novel suite of LLMs specifically trained for multilingual infoveillance on social media platforms. We designed the first multilingual public health infoveillance benchmark, where we evaluated PH-LLM against leading open-source and proprietary LLMs including GPT-4o. The PH-LLM models and associated Python code can be publicly accessible at https://github.com/luoyuanlab/PH-LLM . Methods In this study, we developed a suite of LLMs named PH-LLM, available in six sizes for various computing settings: PH-LLM-0.5B, PH-LLM-1.5B, PH-LLM-3B, PH-LLM-7B, PH-LLM-14B, and PH-LLM-32B. These PH-LLM models were instruction-based fine-tuned on top of Qwen 2.5, 14 using a curated dataset of 593,100 instruction-output pairs based on 30 infoveillance datasets with a total of 96 public health infoveillance tasks and six question-answering datasets. We evaluated the PH-LLM models on 39 tasks with a total of 52,158 instruction-output pairs, across 10 datasets in English, Chinese, Arabic, and Indonesian. Notably, the evaluation datasets were distinct from those used during the development of PH-LLM. The performance of PH-LLM was benchmarked against state-of-the-art instruction-tuned LLMs, including GPT-4o (version 2024-05-13), Llama-3.1-70B-Instruct, Mistral-Large-Instruct-2407, and Qwen2.5-72B-Instruct 14 – 17 . An overview of this study is provided in Figure 1 . Download figure Open in new tab Figure 1. Overview of this study Data Source We conducted a Google search to compile an initial list of publicly available, manually annotated infoveillance datasets based on social media data. Two researchers (XZ and JZ, or XZ and CW) assessed the annotation quality of each dataset. The evaluation criteria included subjective impressions of the study’s quality, the robustness of the annotation process described, and the popularity of the paper and dataset as indicated by metrics such as citations and GitHub stars. Discrepancies between researchers were resolved through discussions. Only datasets that were manually annotated and deemed high quality by both researchers were recruited, either as the training set or the evaluation dataset. We prioritized datasets requiring API access for inclusion in the evaluation set. Datasets annotated using machine learning models were excluded. Ultimately, a total of 40 infoveillance datasets were collected. Of these, 30 datasets, encompassing 96 public health infoveillance tasks concerning vaccine sentiment, hate speech, mental health, NPIs, misinformation, and beyond, were included in the training set. In addition, six additional QA datasets were incorporated to enrich the training corpus. Details of the training set are provided in Supplementary Table 1. The remaining 10 datasets, comprising 39 unseen infoveillance tasks and 52,158 instruction-output pairs, were excluded from model training and reserved for evaluation. Details of the evaluation datasets are presented in Table 1 . Importantly, there was no overlap between the training and evaluation sets. View this table: View inline View popup Table 1. Evaluation benchmark. The evaluation benchmark is based on manually annotated social media datasets. The source of the datasets, language, and the distribution of labels were presented. None of the evaluation datasets were used during the training of PH-LLM models. Note: The prompt templates for the WCV dataset were originally written in Chinese, and their English translations were shown in this table. We construct the prompt templates according to the original data annotation strategy described by the creator of the source datasets without any paraphrasing, whenever possible. Constructing Instruction Datasets Infoveillance Instructions (I 2 ) The I 2 dataset was developed using 30 social media datasets included in the training corpus. Figure 1 illustrated the process of transforming previously annotated social media-based public health infoveillance datasets into instruction datasets, which was applied to create I 2 , an integrated instruction-tuning datasets for training PH-LLM. The datasets in I 2 were sourced from a total of 30 manually annotated social media datasets from prior studies and were either monolingual or multilingual. Detailed information about each training dataset in I 2 is provided in Supplementary Table 1. Typically, social media datasets collected from the Internet contain two primary entries: the social media post (or its ID), and corresponding annotation(s) (e.g., 0 and 1). For datasets containing only post IDs (all sourced from X, formally known as Twitter), we retrieved the actual textual content of the post (excluding replies) via the official X API 18 . Each post was transformed into an instruction comprehensible to humans, and its annotation(s) were converted into the gold-standard response for that instruction. Templates were applied to transform social media posts into instructions, as shown in Figure 1 . To enhance the multilingual capabilities of PH-LLM models, templates for developing the I 2 dataset were translated into 29 languages supported by Qwen 2.5. The list of supported languages is available in Supplementary Material 1. The original social media posts were not translated. Instead, for each post in I 2 , a template in one of the 29 languages was randomly assigned. Each instruction was a combination of one social media post and one template. 17 Templates were initially crafted in Chinese or English, and subsequently translated into 28 additional languages using the web interface of ChatGPT-4o ( https://chatgpt.com/ ). Public Health Question Answering (PHQA) To construct the PHQA dataset, we employed a two-step process. First, we applied a keyword-based filtering approach to extract public health-related instruction-output pairs from three datasets: PubMed summarization, Meadow medical flashcards, and OpenOrca, as a supplement to the training set. 19 – 21 The keywords used for filtering are presented in Supplementary Material 1. Second, we selected subsets from the MedMCQA dataset, focusing specifically on two subjects: Social/Preventive Medicine and Psychiatry. 22 We further sampled 10,000 records from the MentalLLaMA collection, a question-answering dataset regarding mental health derived from gpt-3.5-turbo. 13 We also supplemented the PHQA with a subset of Bactrian-X, 23 a multilingual instruction-output dataset generated by gpt-3.5-turbo, to reinforce the multilingual capabilities of our instruction-tuned model. Merging I 2 and PHQA yielded our training set consisting of 593,100 instruction-output pairs (Supplement Table 1). Evaluation datasets For the evaluation benchmark, we selected 10 high-quality, manually annotated social media datasets ( Table 1 ) that were distinct from the training datasets. We included tasks in these datasets where minority classes comprised at least 5% of the data, as extremely imbalance tasks can cause metric fluctuations and may be less relevant to public health. As illustrated in Figure 1 , we transformed each record in the evaluation datasets into instruction-output pairs based on prompt templates. The original evaluation datasets, prior to the application of instruction templates, were in English, Chinese, Arabic, or Indonesian, whereas the prompt templates for the evaluation datasets were in English or Chinese-English code-mixing ( Table 1 ). Putting datasets and prompt templates together yielded six evaluation datasets with 19 tasks in English and four multilingual evaluation datasets (two in Arabic-English code-mixing, one in Indonesian-English code-mixing, and one in Chinese-English code-mixing), encompassing 20 tasks. In total, we collected 52,158 instruction-output pairs from 39 tasks across 10 datasets in the evaluation benchmark. Detailed prompt templates for each evaluation task are shown in Table 1 . Instruction-tuning of the PH-LLM Models Qwen-2.5, a foundation model developed by Alibaba, was pretrained using up to 18 trillion tokens across more than 29 languages. The model family includes both base model (pretrained only), and instruction models (further trained through instruction-tuning and other methods). 14 We chose the instruction-tuned version of Qwen2.5 as the backbone for the PH-LLM models, given its multilingual capabilities and superior performance in following human instructions 14 . Supporting over 29 languages, Qwen 2.5 enhances the potential applicability of PH-LLM in global health contexts, including low- and middle-income countries (LMICs). To enable efficient LLM finetuning, we utilized quantized low-rank adaption (QLoRA). 24 , 25 For instruction-tuning, our models were trained over 3 epochs, with an effective batch size of 256, a cut-off length of 1024 tokens, a learning rate of 0.00005, incorporating cosine annealing with a warm-up ratio of 0.1, and LoRAPlus learning rate ratio of 16. 26 The instruction-tuning process also adhered to Qwen 2.5’s prompt format to maintain consistency. Elaborations on instruction tuning are available in Supplementary Material 1. Model Evaluation We evaluated the zero-shot performance of PH-LLM models—PH-LLM-0.5B, PH-LLM-1.5B, PH-LLM-3B, PH-LLM-7B, PH-LLM-14B, and PH-LLM-32B—against a wide array of open-source and proprietary LLMs, including GPT-4o (version 2024-05-13), Llama-3.1-72B-Instruct, Mistral-Large-Instruct-2407, and Qwen2.5-72B-Instruct. 14 – 17 During the evaluation of the open-source models (PH-LLM, Llama, Mistral, BLOOMZ, and Qwen 2.5), we used 4-bit quantization with QLoRA to enhance computational efficiency, using gated GPUs servers at Northwestern University. GPT-4o (version 2024-05-13) and GPT-4o mini (version 2024-07-18) were deployed on the Microsoft Azure platform. Statistical Analysis All evaluation datasets focused on classification task, which represent the predominant type of annotated datasets in public health infoveillance. For classification tasks where only one category was relevant to public health, model performance was assessed using the F 1 − score. For tasks involving multiple categories of public health significance, we reported the micro F 1 − score to account for class imbalance. The formulas for calculating precision, recall, F 1 − score, and micro F 1 − score are presented in Supplementary Material 1. Results Table 2 compares the zero-shot performance of PH-LLM on 19 tasks across six English-language datasets against other open-source LLMs of similar sizes. PH-LLM models demonstrated superior average performance, as measured by F 1 − score and micro F 1 − score, compared to their counterparts. Specifically, the smallest model, PH-LLM-0.5B, achieved an average model performance of 30.3%, outperforming Qwen2.5-0.5B-Instruct (23.6%) across 14 of 19 tasks. PH-LLM-1.5B achieved 39.9%, surpassing both Qwen2.5-1.5B-Instruct (36.3%) and the similar-sized Llama-3.2-1B-Instruct (28.0%) on 15 and 16 out of 19 tasks, respectively. View this table: View inline View popup Table 2. Comparison of zero-shot performance on English-language datasets between PH-LLM and other open-source LLMs of similar sizes Among models with ∼7 billion parameters, PH-LLM-7B achieved 48.7%, outperforming bloomz-7b1-mt (27.9%) and Llama-3.1-8B-Instruct (45.4%). However, it performed slightly below Qwen2.5-7B-Instruct (50.7%). PH-LLM-14B (56.0%) consistently outperformed Qwen2.5-14B-Instruct (48.9%) across 13 out of 19 tasks and exceeded Mistral-Nemo-Instruct-2407 (47.1%) on 17 tasks. Remarkably, it also surpassed Mistral-Small-Instruct-2409 (45.8%), which has a larger parameter size of 22 billion. The largest model, PH-LLM-32B, achieved an average performance of 57.9%, surpassing Qwen2.5-32B-Instruct (52.5%). Table 3 presents the zero-shot performance of PH-LLM models on 20 tasks across four multilingual datasets with the same set of benchmark LLMs, where PH-LLM consistently outperformed other models of similar sizes. PH-LLM-0.5B improved upon Qwen2.5-0.5B-Instruct (34.5% vs. 29.5%) on 17 out of 20 tasks. Similarly, PH-LLM-1.5B (42.1%) outperformed both Qwen2.5-1.5B-Instruct (34.1%) and Llama-3.2-1B-Instruct (27.7%), while PH-LLM-3B (48.1%) outperformed both Qwen2.5-3B-Instruct (41.1%) and Llama-3.2-3B-Instruct (40.0%). Among models with∼7 billion parameters, PH-LLM-7B (58.5%) consistently outperformed blooms-7b1-mt (27.3%), as well as Qwen2.5-7B-Instruct (47.4%) and Llama-3.1-8B (47.2%) on most of the 20 tasks. PH-LLM-14B (59.6%) also surpassed Qwen2.5-14B-Instruct (51.5%), Mistral-Small-Instruct-2407 (42.9%), and Mistral-Small-Instruct-2409 (47.4%) in most tasks. PH-LLM-32B achieved an average performance of 61.4%, exceeding Qwen2.5-32B-Instruct’s 55.1%. View this table: View inline View popup Table 3. Comparison of zero-shot performance on multilingual datasets between PH-LLM models and other open-source LLMs of similar sizes Table 4 and Table 5 presents further comparison of PH-LLM models with larger open-source models and proprietary LLMs for both English-language and multilingual datasets. For English-language comparison ( Table 4 ) across 19 tasks, the largest PH-LLM model, PH-LLM-32B (57.9%), demonstrated not only competitive but superior overall performance to other larger open-source models, such as Qwen2.5-72B-Instruct (49.6%) and Llama-3.1-70B-Instruct (52.3%), and Mistral-Large-Instruct-2407 (51.8%). Furthermore, PH-LLM-32B achieved state-of-the-art performance that it outperformed both proprietary LLMs (46.9% of GPT-4o mini and 50.7% of GPT-4o). In Table 5 , for multilingual datasets, PH-LLM-32B continues to outperform all other state-of-the-art models, achieving an average model performance of 61.4% across 20 tasks, specifically Qwen2.5-72B-Instruct (58.5%), Llama-3.1-70B-Instruct (57.7%), Mistral-Large-Instruct-2407 (56.6%), as well as GPT-4o mini (54.1%) and GPT-4o (59.1%). View this table: View inline View popup Table 4. Comparison of zero-shot performance on English-language datasets between PH-LLM-32B and larger open-source models, flagship open-source models, and proprietary LLMs View this table: View inline View popup Table 5. Comparison of zero-shot performance on multilingual datasets between PH-LLM-32B and larger open-source models, flagship open-source models, and proprietary LLMs Figure 2 shows the relationship between average model performance and model size of LLMs evaluated across 19 English evaluation tasks. A positive relationship was observed between the number of parameters in open-source models and their performance. Notably, PH-LLM models demonstrated superior performance compared to models of similar sizes and even larger counterparts. PH-LLM-14B (56.0%) and PH-LLM-32B (57.9%) outperformed strong baselines, including GPT-4o (50.7%), Mistral-Large-Instruct-2407 (51.8%), and Llama-3.1-70B-Instruct (52.3%). Download figure Open in new tab Figure 2. Relationship between model size and average zero-shot performance of LLMs in English evaluation datasets Figure 3 shows the relationship between average model performance and model size across 20 multilingual evaluation tasks. PH-LLM consistently outperformed models of similar sizes and in some cases larger models. PH-LLM-7B (58.5%), in particular, matched the average performance as Qwen2.5-72B-Instruct (58.5%). Moreover, both PH-LLM-14B (59.6%) and PH-LLM-32B (61.4%) surpassed state-of-the-art baseline models, including GPT-4o (59.1%), Qwen2.5-72B-Instruct (58.5%), and GPT-4o mini (54.1%). Download figure Open in new tab Figure 3. Relationship between model size and average zero-shot performance of LLMs in multilingual evaluation datasets for multilingual LLMs officially supporting all languages in the multilingual evaluation Discussion In this study, we introduced PH-LLM, a novel suite of LLMs specialized in public health infoveillance. PH-LLM is available in six model sizes: PH-LLM-0.5B, PH-LLM-1.5B, PH-LLM-3B, PH-LLM-7B, PH-LLM-14B, and PH-LLM-32B. Across diverse public health infoveillance tasks, PH-LLM models consistently demonstrated strong performance, outperforming baseline models of comparable or larger sized in most scenarios. Notably, PH-LLM-14B and PH-LLM-32B achieved superior overall performance on 39 tasks from 10 held-out datasets in public health infoveillance settings, surpassing all baseline models including Llama-3.1-72b-instruct, Mistral-Large-Instruct-2407, Qwen2.5-72b-instruct, and GPT-4o. PH-LLM can reach higher zero-shot performance in public health infoveillance tasks with smaller number of parameters. It reduces the need for extensive GPU resources and complex infrastructure during model deployment and inference, lowering operational costs and making public health infoveillance more accessible, particularly for resource-constrained settings. PH-LLM’s adaptability enables localized and contextualized responses to diverse public health challenges, offering transformative potential for LMICs and other underserved regions. To the best of our knowledge, PH-LLM is the first suite of LLMs specialized in public health infoveillance which is multilingual and publicly available. Previous studies have utilized general-purpose LLMs to advance public health infoveillance on social media platforms, including tasks like data augmentation in social media datasets, 9 , 27 and analyzing public health topics such as vaccine sentiment, mask-wearing behaviors, and mental health. 8 , 10 – 13 , 28 LLMs have also shown potential in assisting public health practice beyond infoveillance, including pandemic forecasting and information extraction. 29 , 30 However, almost all these studies applied general-purpose LLMs like LLaMA and ChatGPT rather than developing LLMs tailored for public health settings, 31 , 32 and they focused predominantly on English-language scenarios. PH-LLM emphasizes multilingual capabilities, extending its utility to non-English contexts, which addresses the diverse linguistic needs of global public health. PH-LLM is designed to be accessible to public health professionals without requiring a background in computer science. With metadata (time, location, social-economic status, and beyond) associated with each social media post, aggregating predictions from PH-LLM can reveal spatiotemporal trends of opinions and behaviors, from nuances on social media platforms, and subsequently underline their public health significance. For example, to inform an HPV vaccination program, public health agencies can apply PH-LLM to stay updated with sudden changes in vaccine acceptance and confidence, trending concerns and misinformation on vaccines, and potential distrust in public health professionals, pharmaceutical companies, or the government. Additionally, tools like LlamaFactory enable users to interact with PH-LLM and effortlessly analyze large-scale data through a user-friendly interface 33 . (Supplementary Figure 1) PH-LLM exhibited strong zero-shot performance for analyzing social media posts relevant to public health. Its performance could be further enhanced potentially through prompt engineering and integration with retrieval-augmented generation and knowledge graph – incorporating contextualized and localized knowledge from public health experts. PH-LLM equips public health systems with a tool to address future emerging infectious diseases and global health challenges. PH-LLM was trained and evaluated using datasets surrounding vaccine hesitancy, mental health, nonadherence to NPIs, hate speech, and misinformation, and similar challenges may re-emerge in future outbreaks and pandemics. 34 The generalizability of LLMs also allows PH-LLM to address new and evolving infoveillance topics with greater flexibility towards variations in geographies, languages, populations, and cultural, social, economic and political contexts, which is an advantage over the pretrain-finetune paradigm. This study has several limitations. First, every LLMs, including PH-LLM, demonstrated suboptimal results in specific tasks. This is because most of the evaluation tasks are imbalanced and could be challenging. Also, we did not optimize prompt templates to ensure fair comparisons and avoid overfitting. Task-specific prompt engineering and evaluations are recommended before deployment of LLMs in zero-shot public health infoveillance. Second, the training set included only 96 infoveillance tasks, which may limit performance of PH-LLM on tasks less represented within the training corpus. Third, PH-LLM’s training datasets were derived from various previous studies, which may reflect inconsistency in annotation quality and potential biases introduced by annotators. Forth, social media data, which underpins PH-LLM’s training and evaluation, represents a biased subset of the population. Predictions based on such data should be interpreted with caution, especially in contexts involving censorship or self-censorship. Lastly, the evaluation focused exclusively on zero-shot performance, and the few-shot and fine-tuning capabilities of PH-LLM remains untested. Despite these limitations, PH-LLM represents a significant enhancement as a novel suite of LLM tailored for public health infoveillance. Its public availability and state-of-the-art performance demonstrate its potential in public health monitoring and evidence-based policymaking, including in LMICs and among at-risk populations. PH-LLM aspires to equip public health agencies at all levels—global, national and local—with the power of AI to promote public health awareness, inform policy and interventions, and address future global health challenges. Data Availability Models and Python code are available on GitHub (https://github.com/luoyuanlab/PH-LLM). Unfortunately, due to the policy of social media platforms, we cannot share data directly. https://github.com/luoyuanlab/PH-LLM CRediT author statement Conceptualization: Xinyu Zhou, Yuan Luo; Methodology: Xinyu Zhou, Yuan Luo, Jiaqi Zhou, Chiyu Wang, Kaize Ding, Qianqian Xie, Yuntian Liu, Zhiyuan Cao, Hua Xu; Software: Xinyu Zhou, Chiyu Wang, Jiaqi Zhou, Huangrui Chu; Validation: Jiaqi Zhou; Formal analysis: Xinyu Zhou; Investigation: Xinyu Zhou; Resources: Yuan Luo, Heidi J. Larson, Xinyu Zhou, Huangrui Chu; Data Curation: Xinyu Zhou, Heidi J. Larson; Writing - Original Draft: Xinyu Zhou; Writing - Review & Editing: Xinyu Zhou, Jiaqi Zhou, Chiyu Wang, Qianqian Xie, Kaize Ding, Chengsheng Mao, Yuntian Liu, Zhiyuan Cao, Huangrui Chu, Xi Chen, Hua Xu, Heidi J. Larson, Yuan Luo; Visualization: Xinyu Zhou, Jiaqi Zhou; Supervision: Yuan Luo; Project administration: Yuan Luo, Xinyu Zhou; Funding acquisition: Yuan Luo; All authors have read and approved the manuscript. Data sharing Models and Python code are available on GitHub ( https://github.com/luoyuanlab/PH-LLM ). Unfortunately, due to the policy of social media platforms, we cannot share data directly. Declaration of interests The authors have declared no competing interest. Acknowledgments This study is supported in part by NIH grants R01LM013337 (YL). Footnotes ↵ * co-first author Reference 1. ↵ Eysenbach G . Infodemiology and infoveillance: framework for an emerging set of public health informatics methods to analyze search, communication and publication behavior on the Internet . Journal of medical Internet research 2009 ; 11 ( 1 ): e1157 . OpenUrl 2. ↵ Calleja N , AbdAllah A , Abad N , et al. A public health research agenda for managing infodemics: methods and results of the first WHO infodemiology conference . JMIR infodemiology 2021 ; 1 ( 1 ): e30979 . OpenUrl CrossRef PubMed 3. Terry K , Yang F , Yao Q , Liu C . The role of social media in public health crises caused by infectious disease: a scoping review . BMJ Global Health 2023 ; 8 ( 12 ): e013515 . OpenUrl Abstract / FREE Full Text 4. Purba AK , Pearce A , Henderson M , McKee M , Katikireddi SV . Social media as a determinant of health . European Journal of Public Health 2024 ; 34 ( 3 ): 425 – 6 . OpenUrl PubMed 5. ↵ Infodemic . https://www.who.int/health-topics/infodemic#tab=tab_1 (accessed December 10 2024 ). 6. ↵ Gunasekeran DV , Tseng RMWW , Tham Y-C , Wong TY. Applications of digital health for public health responses to COVID-19: a systematic scoping review of artificial intelligence, telehealth and related technologies . NPJ digital medicine 2021 ; 4 ( 1 ): 40 . OpenUrl PubMed 7. ↵ Tsao S-F , Chen H , Tisseverasinghe T , Yang Y , Li L , Butt ZA . What social media told us in the time of COVID-19: a scoping review . The Lancet Digital Health 2021 ; 3 ( 3 ): e175 – e94 . OpenUrl 8. ↵ Espinosa L , Salathé M . Use of large language models as a scalable approach to understanding public health discourse . medRxiv 2024 : 2024.02.06.24302383 . 9. ↵ Guo Y , Ovadje A , Al-Garadi MA , Sarker A . Evaluating large language models for health-related text classification tasks with public social media data . Journal of the American Medical Informatics Association 2024 ; 31 ( 10 ): 2181 – 9 . OpenUrl PubMed 10. ↵ He L , Omranian S , McRoy S , Zheng K . Using Large Language Models for sentiment analysis of health-related social media data: empirical evaluation and practical tips . medRxiv 2024 : 2024.03.19.24304544 . 11. Kim S , Kim K , Jo CW . Accuracy of a large language model in distinguishing anti-and pro-vaccination messages on social media: The case of human papillomavirus vaccination . Preventive Medicine Reports 2024 ; 42 : 102723 . OpenUrl PubMed 12. Shah SM , Gillani SA , Baig MSA , Saleem MA , Siddiqui MH . Advancing Depression Detection on Social Media Platforms Through Fine-Tuned Large Language Models . arXiv preprint arXiv : 240914794 2024 . 13. ↵ Yang K , Zhang T , Kuang Z , Xie Q , Huang J , Ananiadou S . MentaLLaMA: interpretable mental health analysis on social media with large language models . Proceedings of the ACM on Web Conference 2024 ; 2024 ; 2024 . p. 4489 – 500 . OpenUrl 14. ↵ Yang A , Yang B , Hui B , et al. Qwen2 technical report . arXiv preprint arXiv : 240710671 2024 . 15. Dubey A , Jauhri A , Pandey A , et al. The llama 3 herd of models . arXiv preprint arXiv : 240721783 2024 . 16. Large Enough | Mistral AI | Frontier AI in your hands . 2024 . https://mistral.ai/news/mistral-large-2407/ (accessed October 17 2024 ). 17. ↵ Muennighoff N , Wang T , Sutawika L , et al. Crosslingual generalization through multitask finetuning . arXiv preprint arXiv : 221101786 2022 . 18. ↵ X API | Products - Twitter Developer Platform . https://developer.x.com/en/products/x-api (accessed October 17 2024 ). 19. ↵ White J . PubMed 2.0 . Medical reference services quarterly 2020 ; 39 ( 4 ): 382 – 7 . OpenUrl CrossRef PubMed 20. Mukherjee S , Mitra A , Jawahar G , Agarwal S , Palangi H , Awadallah A . Orca: Progressive learning from complex explanation traces of gpt-4 . arXiv preprint arXiv : 230602707 2023 . 21. ↵ Han T , Adams LC , Papaioannou J-M , et al. MedAlpaca--an open-source collection of medical conversational AI models and training data . arXiv preprint arXiv : 230408247 2023 . 22. ↵ Pal A , Umapathi LK , Sankarasubbu M. Medmcqa: A large-scale multi-subject multi-choice dataset for medical domain question answering . Conference on health, inference, and learning; 2022: PMLR ; 2022 . p. 248 – 60 . 23. ↵ Li H , Koto F , Wu M , Aji AF , Baldwin T . Bactrian-x: Multilingual replicable instruction-following models with low-rank adaptation . arXiv preprint arXiv : 230515011 2023 . 24. ↵ Dettmers T , Pagnoni A , Holtzman A , Zettlemoyer L . Qlora: Efficient finetuning of quantized llms . Advances in Neural Information Processing Systems 2024 ; 36 . 25. ↵ Hu EJ , Shen Y , Wallis P , et al. Lora: Low-rank adaptation of large language models . arXiv preprint arXiv : 210609685 2021 . 26. ↵ Hayou S , Ghosh N , Yu B . Lora+: Efficient low rank adaptation of large models . arXiv preprint arXiv : 240212354 2024 . 27. ↵ Jiang Y , Qiu R , Zhang Y , Zhang P-F . Balanced and explainable social media analysis for public health with large language models . Australasian Database Conference ; 2023 : Springer; 2023 . p. 73 – 86 . 28. ↵ Li W , Zhu Y , Lin X , Li M , Jiang Z , Zeng Z . Zero-shot Explainable Mental Health Analysis on Social Media by Incorporating Mental Scales . Companion Proceedings of the ACM on Web Conference 2024 ; 2024 ; 2024 . p. 959 – 62 . OpenUrl 29. ↵ Du H , Zhao J , Zhao Y , et al. Advancing Real-time Pandemic Forecasting Using Large Language Models: A COVID-19 Case Study . arXiv preprint arXiv : 240406962 2024 . 30. ↵ Harris J , Laurence T , Loman L , et al. Evaluating Large Language Models for Public Health Classification and Extraction Tasks . arXiv preprint arXiv : 240514766 2024 . 31. ↵ Touvron H , Lavril T , Izacard G , et al. Llama: Open and efficient foundation language models . arXiv preprint arXiv : 230213971 2023 . 32. ↵ Achiam J , Adler S , Agarwal S , et al. Gpt-4 technical report . arXiv preprint arXiv : 230308774 2023 . 33. ↵ Zheng Y , Zhang R , Zhang J , et al. Llamafactory: Unified efficient fine-tuning of 100+ language models . arXiv preprint arXiv : 240313372 2024 . 34. ↵ Depoux A , Martin S , Karafillakis E , Preet R , Wilder-Smith A , Larson H. The pandemic of social media panic travels faster than the COVID-19 outbreak . Oxford University Press ; 2020 . p. taaa031 . 35. Poddar S , Samad AM , Mukherjee R , Ganguly N , Ghosh S . Caves: A dataset to facilitate explainable classification and summarization of concerns towards covid vaccines . Proceedings of the 45th international ACM SIGIR conference on research and development in information retrieval; 2022 ; 2022 . p. 3154 – 64 . 36. Müller M , Salathé M , Kummervold PE . Covid-twitter-bert: A natural language processing model to analyse covid-19 content on twitter . Frontiers in artificial intelligence 2023 ; 6 : 1023281 . OpenUrl PubMed 37. Mollas I , Chrysopoulou Z , Karlos S , Tsoumakas G . ETHOS: a multi-label hate speech detection dataset . Complex & Intelligent Systems 2022 ; 8 ( 6 ): 4663 – 78 . OpenUrl 38. Kennedy B , Atari M , Davani AM , et al. Introducing the Gab Hate Corpus: defining and applying hate-based rhetoric to social media posts at scale . Language Resources and Evaluation 2022 : 1 – 30 . 39. Memon SA , Carley KM . Characterizing covid-19 misinformation communities using a novel twitter dataset . arXiv preprint arXiv : 200800791 2020. 40. Lin L , Song Y , Wang Q , et al. Public attitudes and factors of COVID-19 testing hesitancy in the United Kingdom and China: comparative infodemiology study . JMİR infodemiology 2021 ; 1 ( 1 ): e26895 . OpenUrl 41. Ameur MSH , Aliane H . AraCOVID19-MFH: Arabic COVID-19 multi-label fake news & hate speech detection dataset . Procedia Computer Science 2021 ; 189 : 232 – 41 . OpenUrl 42. Saputri MS , Mahendra R , Adriani M. Emotion classification on indonesian twitter dataset . 2018 International Conference on Asian Language Processing (IALP) ; 2018 : IEEE; 2018 . p. 90 – 5 . 43. Alqurashi S , Hamoui B , Alashaikh A , Alhindi A , Alanazi E . Eating garlic prevents COVID-19 infection: Detecting misinformation on the Arabic content of Twitter . arXiv preprint arXiv : 210105626 2021 . 44. Hou Z , Tong Y , Du F , et al. Assessing COVID-19 vaccine hesitancy, confidence, and public engagement: a global social listening study . Journal of medical Internet research 2021 ; 23 ( 6 ): e27632 . OpenUrl PubMed View the discussion thread. Back to top Previous Next Posted February 10, 2025. Download PDF Supplementary Material Data/Code Email Thank you for your interest in spreading the word about medRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following PH-LLM: Public Health Large Language Models for Infoveillance Message Subject (Your Name) has forwarded a page to you from medRxiv Message Body (Your Name) thought you would like to see this page from the medRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share PH-LLM: Public Health Large Language Models for Infoveillance Xinyu Zhou , Jiaqi Zhou , Chiyu Wang , Qianqian Xie , Kaize Ding , Chengsheng Mao , Yuntian Liu , Zhiyuan Cao , Huangrui Chu , Xi Chen , Hua Xu , Heidi J. Larson , Yuan Luo medRxiv 2025.02.08.25321587; doi: https://doi.org/10.1101/2025.02.08.25321587 Share This Article: Copy Citation Tools PH-LLM: Public Health Large Language Models for Infoveillance Xinyu Zhou , Jiaqi Zhou , Chiyu Wang , Qianqian Xie , Kaize Ding , Chengsheng Mao , Yuntian Liu , Zhiyuan Cao , Huangrui Chu , Xi Chen , Hua Xu , Heidi J. Larson , Yuan Luo medRxiv 2025.02.08.25321587; doi: https://doi.org/10.1101/2025.02.08.25321587 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Health Informatics Subject Areas All Articles Addiction Medicine (570) Allergy and Immunology (864) Anesthesia (302) Cardiovascular Medicine (4445) Dentistry and Oral Medicine (444) Dermatology (383) Emergency Medicine (609) Endocrinology (including Diabetes Mellitus and Metabolic Disease) (1515) Epidemiology (15236) Forensic Medicine (30) Gastroenterology (1127) Genetic and Genomic Medicine (6610) Geriatric Medicine (669) Health Economics (1000) Health Informatics (4549) Health Policy (1370) Health Systems and Quality Improvement (1613) Hematology (543) HIV/AIDS (1266) Infectious Diseases (except HIV/AIDS) (15926) Intensive Care and Critical Care Medicine (1104) Medical Education (623) Medical Ethics (147) Nephrology (668) Neurology (6613) Nursing (346) Nutrition (999) Obstetrics and Gynecology (1147) Occupational and Environmental Health (957) Oncology (3341) Ophthalmology (975) Orthopedics (369) Otolaryngology (420) Pain Medicine (436) Palliative Medicine (130) Pathology (665) Pediatrics (1694) Pharmacology and Therapeutics (693) Primary Care Research (714) Psychiatry and Clinical Psychology (5458) Public and Global Health (9244) Radiology and Imaging (2205) Rehabilitation Medicine and Physical Therapy (1370) Respiratory Medicine (1197) Rheumatology (596) Sexual and Reproductive Health (715) Sports Medicine (530) Surgery (713) Toxicology (99) Transplantation (289) Urology (265) (function(){function c(){var b=a.contentDocument||a.contentWindow.document;if(b){var d=b.createElement('script');d.innerHTML="window.__CF$cv$params={r:'a025a07e8b88dfa9',t:'MTc3OTg5MTI3Ng=='};var a=document.createElement('script');a.src='/cdn-cgi/challenge-platform/scripts/jsd/main.js';document.getElementsByTagName('head')[0].appendChild(a);";b.getElementsByTagName('head')[0].appendChild(d)}}if(document.body){var a=document.createElement('iframe');a.height=1;a.width=1;a.style.position='absolute';a.style.top=0;a.style.left=0;a.style.border='none';a.style.visibility='hidden';document.body.appendChild(a);if('loading'!==document.readyState)c();else if(window.addEventListener)document.addEventListener('DOMContentLoaded',c);else{var e=document.onreadystatechange||function(){};document.onreadystatechange=function(b){e(b);'loading'!==document.readyState&&(document.onreadystatechange=e,c())}}}})();
Text is read by the "Ask this paper" AI Q&A widget below.
Extraction quality varies by source — PMC NXML preserves structure
cleanly, OA-HTML may include some navigation residue, and OA-PDF can
have broken hyphenation. The publisher copy
(via DOI)
is the canonical version.