Full text
41,984 characters
· extracted from
preprint-html
· click to expand
A machine learning model to support the screening for methods guidance articles in MEDLINE: A performance evaluation of ASReview simulation mode | medRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-P4HH5NV'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search A machine learning model to support the screening for methods guidance articles in MEDLINE: A performance evaluation of ASReview simulation mode View ORCID Profile Wael Abdelkader , View ORCID Profile Daniel Xie , Cynthia Lokker , Lingyang Chu , Stefan Schandelmaier , Ashirbani Saha , Muhammad Afzal , Alfonso Iorio doi: https://doi.org/10.1101/2025.10.13.25337935 Wael Abdelkader 1 Department of Health Research Methods, Evidence, and Impact, McMaster University , Canada Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Wael Abdelkader For correspondence: Abdelkaw{at}mcmaster.ca Daniel Xie 3 Department of Integrated Biomedical Engineering and Health Sciences, McMaster University , Canada Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Daniel Xie Cynthia Lokker 1 Department of Health Research Methods, Evidence, and Impact, McMaster University , Canada Find this author on Google Scholar Find this author on PubMed Search for this author on this site Lingyang Chu 2 Department of Computing and Software, McMaster University , Canada Find this author on Google Scholar Find this author on PubMed Search for this author on this site Stefan Schandelmaier 4 Division of Clinical Epidemiology, Department of Clinical Research, University of Basel , Switzerland Find this author on Google Scholar Find this author on PubMed Search for this author on this site Ashirbani Saha 1 Department of Health Research Methods, Evidence, and Impact, McMaster University , Canada 5 Department of Oncology, Faculty of Health Sciences, McMaster University , Canada Find this author on Google Scholar Find this author on PubMed Search for this author on this site Muhammad Afzal 1 Department of Health Research Methods, Evidence, and Impact, McMaster University , Canada 6 Department of Computer Science, Birmingham City University , England, United Kingdom Find this author on Google Scholar Find this author on PubMed Search for this author on this site Alfonso Iorio 1 Department of Health Research Methods, Evidence, and Impact, McMaster University , Canada Find this author on Google Scholar Find this author on PubMed Search for this author on this site Abstract Full Text Info/History Metrics Data/Code Preview PDF Abstract Background Advances in clinical research methods are frequently published in biomedical journals, but identifying these articles remains challenging due to their rapid growth and insufficient indexing in biomedical databases. These challenges hinder the curation of methodologically focused resources like the Library of Guidance for Health Scientists (LIGHTS). Traditional screening approaches, such as Boolean search strategies and manual abstract screening, are inefficient and resource-intensive, limiting the feasibility of regularly updating LIGHTS. Machine learning (ML), particularly active learning models, presents a promising solution to improve the efficiency of article screening. Objectives This study evaluates the performance of ASReview’s active learning feature in identifying relevant methods guidance articles using pre-labeled data in simulation mode. Methods Using pre-labeled dataset composed of 1500 methods guidance articles and 20000 clinical studies, categorized as relevant or irrelevant, we trained and compared multiple simulation models in ASReview using various classifiers and feature extraction models. These included combinations of Support Vector Machine (SVM), Naïve Bayes (NB), Neural Network with Sentence BERT (sBERT), Doc2Vec, and TF-IDF. Model performance was evaluated based on screening burden, recall, Work Saved over Sampling (WSS), and precision. All model combinations used maximum query and dynamic double sampling settings. Results At 95-99.5% recall, SVM with TF-IDF required the fewest screened records (6.87-7.66% burden), while SVM with Doc2Vec achieved the best overall performance at 100% recall with only 11.47% screening burden (WSS@100 = 88.5%) in 42 minutes. Models using sBERT for feature extraction performed comparably through 99.5% recall but exhibited severe performance degradation at 100% recall, requiring screening of over 65% of the corpus. Conclusion Classical feature extraction methods, TF-IDF and Doc2Vec, paired with SVM outperform deep learning embeddings methods. ASReview in this controlled setting is a feasible tool for screening methodological literature. Future work should include prospective, human-in-the-loop experiments that embed the Doc2Vec-based SVM pipeline in comparison to human screening. Introduction Methods guidance articles are a primary resource for health researchers and are essential to minimize research inefficiencies and avoid research waste. These publications offer advice on designing, conducting, analyzing, interpreting, reporting, or assessing health research studies 1 . Leading methodologists regularly publish methods guidance articles in top medical journals but also in specialized clinical and methodological journals. Identifying these articles in general biomedical databases such as MEDLINE can be very challenging due to the lack of validated search filters to separate methodological from non-methodological articles, insufficient indexing terms and indexing of methods papers, and highly variable reporting standards 2 . This is exacerbated by the exponential growth in published biomedical literature at a rate of at least one new article every 26 seconds 3 . In 2022, over 1.3 million new citations were indexed in MEDLINE, almost double the citations added 5 years prior 4 . The Library of Guidance for Health Scientists (LIGHTS, www.lights.science ) is a new living online repository of methods guidance articles designed to improve access to methods focused articles for health researchers 2 . LIGHTS is currently maintained through regular manual screening of selected journals, journal methods sections, and series in selected journals, as well as selected social media accounts and recommendations from researchers which are limited sources for methods guidance articles. A formal Boolean search strategy combined with human screening was piloted and proved not to be feasible due to the very large number of records to be screened 2 . To improve the comprehensiveness and efficiency of updating LIGHTS, a feasible screening procedure is needed. ML falls within the realm of artificial intelligence 5 . It is defined as a range of mathematical techniques designed to autonomously recognize patterns in data and subsequently use these patterns to predict future data or other relevant outcomes 6 , 7 . In healthcare, machine learning models can effectively identify patterns and relationships, leading to accurate predictions, e.g., patient outcomes and diagnosis 8 , 9 . In the biomedical literature field, a primary use of machine learning is text mining—a computer-driven process that extracts novel insights from textual data 10 , 11 . Example model functions include classifying studies and recognizing high-quality evidence 12 , and easing the burden of systematic evidence synthesis by reducing screening workload 13 – 16 . Active learning, also known as "query learning," is a subset of machine learning with the central premise of permitting the learning algorithm to actively choose the data it learns from, displaying a sort of curiosity, to achieve superior performance with a reduced amount of training data 17 . Active learning is a specialized form of supervised machine learning that allows an algorithm to interactively query a user or another information source to label new data points with desired outputs 17 , 18 . Instead of requiring labels for every instance upfront, it begins with a small, labeled dataset and actively selects additional data points to be labeled 17 , 19 , 20 . This approach aims to construct a high-performance classifier while minimizing the training dataset size. The typical process involves starting with a small, labeled dataset, training an initial model, estimating prediction uncertainty, selecting instances with low confidence using various query strategies, obtaining labels from an oracle, updating the model with newly annotated data, and repeating the steps until a performance threshold is reached 17 , 18 , 20 – 22 . The trained model when applied to unseen data ranks the instances based on their relevance, prioritizing those that are most relevant 17 . In the context of biomedical literature, the ranking by the model potentially decreases the screening workload by allowing the quicker identification of all relevant documents early, without the need to manually review the complete dataset 17 , 20 . Certain groups that maintain living databases for specific methodological areas, such as recruitment methods (Online Resource for Recruitment research in Clinical triAls, ORRCA) and core outcome sets (Core Outcome Measures in Effectiveness Trials, COMET), have successfully implemented learn-to-rank models to enhance their screening processes 23 – 25 . Whether such models perform well in a more broadly defined methodological context such as “methods guidance” remains unknown. Objective Using ASReview in simulation mode, our objective is to evaluate the ability of the active learning platform, ASReview, to identify methods guidance articles in MEDLINE. We aimed to compare multiple models’ configurations in terms of screening burden, precision, WSS, and elapsed time at recall levels up to 100 % to identify the most efficient model. Methods ASReview The ASReview platform version 1.6.6 was used for developing the active screening models 20 . ASReview is an open-source software that leverages active learning for model training to enhance the systematic literature review screening process. In this context, active learning involves a ML model working in conjunction with a human reviewer to iteratively select the most pertinent documents from a vast corpus of unreviewed literature 20 . ASReview platform encompasses various ML classifiers such as logistic regression, Naïve Bayes (NB), Support Vector Machine (SVM) and Convolutional Neural Network (CNN); and feature extraction methods, such as Sentence Bidirectional Encoder Representations from Transformers (sBERT), word embedding, Term-Frequency Inverse Document Frequency (TF/IDF), and Doc2Vec. Using the provided training set, these algorithms analyze patterns in the labeled articles, gaining insights into the characteristics that make an article relevant to the research topic 20 . To train active learning screening models, we usually require a small set of labelled articles, which provides an initial training seed for the model and can be as minimal as one relevant and one irrelevant article 20 . Following this, the ML model selects unlabelled articles with the least certainty and presents them to a human reviewer for assessment. In this iterative process, the human reviewer labels these articles as relevant or irrelevant. For this study, we are utilizing ASReview in simulation mode, in which ASReview’s re-plays the entire active-learning workflow on a fully labelled corpus. First, the fully labelled dataset is loaded with a small “prior knowledge” seed. Typically, a handful of relevant and irrelevant articles are supplied to initialize the model. The chosen classifier and feature extraction methods are then trained on this seed; the model ranks the remaining unseen records by predicted relevance. Rather than awaiting human judgment for each instance, ASReview immediately consults the concealed pre-labelled data, updates the model with that information, and iterates this loop at machine speed simulating human screening. Throughout the run ASReview logs the order in which each record is screened, the cumulative recall after every iteration, and the number of records examined, allowing for recall, burden, and WSS calculations at different thresholds. Simulation mode produces perfectly reproducible performance traces for any classifier-feature combination, enabling rapid benchmarking before prospective deployment on unlabeled literature and human screening 20 . Study design We conducted a simulation-based evaluation of 6 ASReview classifier–feature combinations. Performance was recorded at 95%, 98%, 98.5%, 99%, 99.5%, and 100% recall thresholds. Active-screening performance metrics were reported, including WSS, precision, Burden, and time to achieve 100% recall 26 , 27 ( supplementary Table 1 ). We used 6 different configurations for the models: NB and TF/IDF, SVM and TF/IDF, SVM and Doc2Vec, SVM and sBERT, CNN and Doc2Vec, and CNN and sBERT. We used ten articles as prior knowledge, of which five were relevant and five were irrelevant. The articles used as prior knowledge were fixed across all models and were selected by a random number generator (supplementary Table 3). We used the maximum query strategy combined with a double dynamic resampling balancing strategy for all experiments. All experiments were conducted on a single machine with a 13-generation core i9-13900H CPU, 32Gb of RAM, and Nvidia RTX-4070 GPU. Dataset description We used a pre-labelled dataset consisting of 21,535 PubMed articles, of which 1,535 articles were retrieved from the LIGHTS database, and they represented the “relevant” class 2 . Articles are included in the LIGHTS database if they meet the following inclusion criteria at full-text screening: Published in a peer-reviewed journal. Title, series title, section title, author keywords, or abstract states aim to provide guidance on a methodological topic. Different terms for guidance are accepted, such as tutorial, recommendation, best practice, among others. The articles address methods relevant to health-related research involving humans. The remaining 20,000 articles representing the “irrelevant” class were retrieved from the McMaster PLUS database which excludes methods guidance articles as one of its inclusion criteria. The McMaster PLUS dataset includes titles and abstracts of 155,679 articles published between 2012 and 2024. These articles were identified using PubMed identifiers developed by the Health Research Information Unit (HiRU) at McMaster University and were labeled by expert research associates 28 , 29 . Results The number of articles screened by the different classifier feature combinations at the various recall thresholds for relevant articles (95%, 98%, 98.5%, 99%, 99.5%, and 100%) are shown in Table 1 . View this table: View inline View popup Table 1: Screening performance for models tested configurations at defined recall thresholds in the ASReview simulation. At the 95% recall threshold, the SVM paired with TF-IDF required the fewest screened records (n=1480). The next best configurations were sBERT combined with SVM classifier and CNN with each requiring 1500 and 1510 records, respectively. These three models remained the top three performing models till the 99.5% recall threshold with the SVM+TF/IDF remaining the best performing while the two sBERT models alternated between second and third place separated by only a small margin. At 100% recall, the two SVM-based models with classical feature extraction set the efficiency benchmark with SVM + Doc2Vec reached 100% recall after reviewing only 2470 records, 11.47% burden, yielding a WSS@100 of 88.5%. Precision decreased from 94.7% at 95% recall to almost 62% at 100% recall, however, the SVM+Doc2Vec model completed the task in just 42 minutes. SVM + TF-IDF was the second-best performing mode, with a marginally lower WSS@100 of 86.5% and a longer time of 15 hours 31 minutes; nevertheless, its precision remained the highest below 100 % recall ( Table 2 ). View this table: View inline View popup Table 2: Models’ screening burden, precision, and WSS performance metrics at recall thresholds. By contrast, models using sBERT for feature extraction recorded higher burden and runtime once recall surpassed the 99% threshold. SVM + sBERT and NN + sBERT had to screen more than 14,000 and 13,900 records, respectively, representing over 65% of the corpus, driving precision below 35% and WSS@100 to near-random levels, all while consuming 33 to 45 hours of compute time. The Naïve Bayes model with TF-IDF sat between these extremes needing only a one-minute runtime to reach 100% recall at the price of a higher screening burden of 18.3% and reduced WSS of 81.8% ( Table 2 ). Discussion In this study, we used ASReview simulation engine combined with a fully labelled corpus of 1535 methods guidance articles and 20000 non-guidance citations to evaluate the performance of the assisted screening as a component to reduce the screening workload towards the identification of methods guidance articles. Six classifier-feature extraction configurations were tested at recall thresholds ranging from 95% to 100%. At 95% recall, all models had a similar performance with WSS@95 around 92% with less computationally demanding models achieving the highest performance. This indicates that ASReview can effectively reduce the workload of methods guidance articles screening. At higher recall thresholds up to 99.5% recall, performance was still comparable with the SVM models achieving the best workload reduction indicating negligible benefit from deep feature extraction in this range. At 100% recall threshold, deep feature extraction methods tended to underperform in both the amount of workload burden needed to reach 100% of almost 70% of the dataset, with much longer computational demands in comparison to lightweight models like NB and SVM models. In the study done by Ferdinands et al. (2023), they used multiple models in ASReview simulation and concluded that lightweight models can outperform complex computationally demanding models in terms of WSS@95 and in the time elapsed to complete the screening 30 , 31 . While screening tasks conventionally aim for 100% recall, ensuring all relevant articles are identified, this endpoint offers limited insight into the comparative performance of assisted screening methods. The meaningful distinction between approaches, particularly those employing ML assisted screening, emerges in how efficiently they reduce the screening workload and burden while navigating the most uncertain articles during the screening process. This zone of uncertainty is not fixed but rather dataset-dependent; in one review, the most challenging decisions may cluster around 99% recall as in our case, while in another, ambiguity may peak much earlier at 80% recall due to differences in topic complexity, terminology consistency, or any other reason. Therefore, performance evaluation should focus not on the inevitable endpoint of complete recall, but on how effectively different methods handle uncertainty at whatever recall threshold it occurs. To date, no experiments have applied ML to identify methods guidance articles 12 , 32 . Our examination of ASReview offers an approach offering directions for the integration of ML into the LIGHTS database infrastructure, facilitating the otherwise unmanageable screening of results from biomedical database searches. One of our study strengths is the use of a large, expertly labelled reference standard encompassing both a substantial positive class of methods guidance articles in combination of a large pool of 20,000 irrelevant citations, thereby allowing precise estimates of burden, precision and WSS 31 , 33 . The use of the simulation mode is advocated for evaluating active learning models before systematic screening 34 . The use of ASReview’s simulation with a fully labelled corpus eliminates the inter-reviewer variability, thereby enabling the evaluation of multiple models’ configurations. Under identical conditions, a reproducible head-to-head comparison can be achieved by isolating the models’ performance from reviewer variability 34 . Furthermore, comparing classifier–feature pairs is an important step towards trustworthy screening automation as screening efficiency at a fixed recall depend not only on the classifier and active learning strategy used but also on how documents are represented. Prior studies show that different classifier-feature choices can affect WSS and screening burden at high recall even under the same dataset and labeling strategy (Ferdinands et al., 2023). ASReview offers multiple algorithms and vectorizers but does not prescribe a default best, making empirical head-to-head testing necessary for transparent, reproducible selection in each review (van de Schoot et al., 2021). Modern embedding models such as sBERT for example capture semantics that bag-of-words or TF-IDF representations may miss, plausibly shifting prioritization curves (Reimers & Gurevych, 2019). This comparison provides evidence directly relevant to resource constrained groups considering whether the computational overhead of deep embeddings is justified over less computationally demanding approaches such as TF-IDF or Doc2Vec embedding (Bailly et al., 2022). Finally, the use of fixed seeds and a publicly available open-source platform ensures that our workflow is fully replicable and readily extendable to other biomedical corpora, thereby enhancing both the transparency and generalizability of our findings. The main limitation is the generalizability of our results given the selective dataset combining methods guidance articles with medical studies. The dataset provided a high contrast between relevant and irrelevant whereas the results of real Boolean searches would typically contain a substantial proportion of difficult-to-classify articles such as methods papers that do not satisfy the criteria for “methods guidance.” Another limitation is the size of the dataset used. Although the 21,535 records dataset is sizeable by ML standards, it does not approximate the scale encountered in real-world methods guidance searches. Hirt et al. found that a keyword-based MEDLINE query for methodological guidance returned more than 915,000 citations 2 . Consequently, the workload reductions observed here are considered an optimistic step towards the incorporation of ML assisted screening within the LIGHTS database workflow. Our findings suggest a practical path towards future adoption into the LIGHTS database screening process which includes three main steps: model selection, validation, and implementation. Using the simulation mode in ASReview in this study allowed us to select the best performing classifier-feature pair to maximize WSS at high recalls while minimizing screening burden and compute time. Following we aim to validate the selected model using a prospective, human-in-the-loop validation on an unlabeled, mixed-topic dataset to confirm that the selected model achieves the simulated WSS, burden reduction, and precision at desired recall. Conclusion Our study results show that combining an SVM with a lightweight feature representation, Doc2Vec, offers the most favourable balance of low screening burden, high WSS@95 to WSS@100, high precision, and minimal elapsed time, whereas deep-embedding architectures confer little advantage in this simulation and can dramatically inflate workload and turnaround. Future work should include a more real-world representative dataset and prospective, human-in-the-loop experiments that embed the Doc2Vec-based SVM pipeline in comparison to human screening. Future experiments should also explore different stopping criteria driven by informed decisions to balance the risk of missing any relevant articles and the costs of continued screening and its burden. Data Availability The data underlying the results presented in the study are available from McMaster PLUS: https://plus.mcmaster.ca/mcmasterplusdb/ LIGHTS database: https://lights.science/ Supplementary information Search strategy for MEDLINE / Ovid Below is the MEDLINE search strategy that the LIGHTS team developed. Search date: 28/07/2021; Ovid MEDLINE(R) ALL years 2010 to 2020. Number of hits: 876,395 . Because the number of hits is so large, current search updates of LIGHTS are based on a version of this strategy that is limited to the most relevant journals only. (guideline OR practice guideline OR consensus development conference OR consensus development conference, NIH).pt. OR Guidelines as Topic/ OR practice guideline/ OR Checklist/ OR exp standards/ OR exp Consensus/ OR exp Consensus Development Conferences as Topic/ OR exp Delphi Technique/ OR Decision Making/ OR (guideline OR guidance OR guide OR guiding OR Tutorial OR Tutorials OR white paper OR Framework OR Checklist OR Checklists OR step-by-step OR Primer OR pitfall OR Pitfalls OR consensus* OR Delphi OR Expert-panel).ti,ab,kf,kw. OR ((provide OR providing OR provided OR provision OR give OR giving OR gave OR given OR practical) ADJ2 (advice OR recommend* OR tip OR tips)).ti,ab,kf,kw. OR (Recommend*).ab. /freq=2 OR ((best OR code OR good) ADJ2 (practice OR practices)).ti,ab,kf,kw. OR (guidelines OR Standard OR Standards OR Recommend* OR Elaboration OR elaborating OR Explanation OR explaining OR extension).ti,kf,kw. OR (Statement* OR Principle OR Principles OR Principled OR tool OR tools OR Rule OR Rules OR how-to OR critical-question*).ti. AND (201*.ez,ep,dt. OR 201*.dp. OR 2020.ez,ep,dt. OR 2020.dp.) View this table: View inline View popup Supplementary Table 1: Summary of the performance metrics will be used View this table: View inline View popup Supplementary table 2: Prior knowledge studies used as fixed seed. Download figure Open in new tab Supplementary figure 1: Models’ Precision (%) against Recall (%) Download figure Open in new tab Supplementary figure 2: Models’ Burden (%) against Recall (%) Download figure Open in new tab Supplementary figure 3: Models’ WSS (%) against Recall (%) Download figure Open in new tab Supplementary figure 4: Models’ elapsed screening time (%) against Recall (%) References 1. ↵ Hirt J , Ewald H , Lawson DO , Hemkens LG , Briel M , Schandelmaier S . A systematic survey of methods guidance suggests areas for improvement regarding access, development, and transparency . Vol. 149 , J ournal of Clinical Epidemiology . Elsevier Inc .; 2022 . p. 217 – 26 . OpenUrl 2. ↵ Hirt J , Schönenberger CM , Ewald H , Lawson DO , Papola D , Rohner R , et al. Introducing the Library of Guidance for Health Scientists (LIGHTS): A Living Database for Methods Guidance . JAMA Netw Open . 2023 Feb 1; 6 ( 2 ): e2253198 . OpenUrl 3. ↵ Garba S , Ahmed A , Mai A , Makama G , Odigie V . Proliferations of scientific medical journals: a burden or a blessing . Oman Med J [Internet ]. 2010 Oct; 25 ( 4 ): 311 – 4 . Available from: https://pubmed.ncbi.nlm.nih.gov/22043365 OpenUrl 4. ↵ National Library of Medicine . MEDLINE PubMed Production Statistics . 2022 . 5. ↵ Shailaja K , Seetharamulu B , Jabbar MA. Machine Learning in Healthcare: A Review. In: 2018 Second International Conference on Electronics, Communication and Aerospace Technology (ICECA) [Internet] . IEEE ; 2018 . p. 910 – 4 . Available from: https://ieeexplore.ieee.org/document/8474918/ 6. ↵ Murphy KP . Machine Learning A Probabilistic Perspective . MIT Press ; 2012 . 1104 p. 7. ↵ Burkov A . The Hundred-Page Machine Learning Book . 1st ed . Andriy Burkov ; 2019 . 8. ↵ Beam AL , Kohane IS . Big Data and Machine Learning in Health Care . JAMA [Internet ]. 2018 Apr 3; 319 ( 13 ): 1317 . Available from: http://jama.jamanetwork.com/article.aspx?doi=10.1001/jama.2017.18391 OpenUrl 9. ↵ Wiens J , Shenoy ES . Machine Learning for Healthcare: On the Verge of a Major Shift in Healthcare Epidemiology . Clinical Infectious Diseases [Internet ]. 2018 Jan 6; 66 ( 1 ): 149 – 53 . Available from: https://academic.oup.com/cid/article/66/1/149/4085880 OpenUrl 10. ↵ Sumathy KL , Chidambaram M . Text mining: concepts, applications, tools and issues-an overview . Int J Comput Appl . 2013 ; 80 ( 4 ). 11. ↵ Gupta V , Lehal GS . A survey of text mining techniques and applications . Journal of emerging technologies in web intelligence . 2009 ; 1 ( 1 ): 60 – 76 . OpenUrl 12. ↵ Abdelkader W , Navarro T , Parrish R , Cotoi C , Germini F , Iorio A , et al. Machine Learning Approaches to Retrieve High-Quality, Clinically Relevant Evidence From the Biomedical Literature: Systematic Review . JMIR Med Inform [Internet] . 2021 Sep 9; 9 ( 9 ): e30401 . Available from: https://medinform.jmir.org/2021/9/e30401 OpenUrl 13. ↵ Tsou AY , Treadwell JR , Erinoff E , Schoelles K . Machine learning for screening prioritization in systematic reviews: comparative performance of Abstrackr and EPPI-Reviewer . Syst Rev [Internet] . 2020 Dec 2; 9 ( 1 ): 73 . Available from: https://systematicreviewsjournal.biomedcentral.com/articles/10.1186/s13643-020-01324-7 OpenUrl 14. Qin X , Liu J , Wang Y , Liu Y , Deng K , Ma Y , et al. Natural language processing was effective in assisting rapid title and abstract screening when updating systematic reviews . J Clin Epidemiol [Internet] . 2021 ; 133 : 121 – 9 . Available from: http://www.elsevier.com/locate/jclinepi OpenUrl 15. Hashimoto K , Kontonatsios G , Miwa M , Ananiadou S . Topic detection using paragraph vectors to support active learning in systematic reviews . J Biomed Inform [Internet ]. 2016 Aug; 62 : 59 – 65 . Available from: https://linkinghub.elsevier.com/retrieve/pii/S1532046416300442 OpenUrl 16. ↵ Gates A , Johnson C , Hartling L. Technology-assisted title and abstract screening for systematic reviews: a retrospective evaluation of the Abstrackr machine learning tool . Syst Rev . 2018 Mar;7. 17. ↵ Settles B. Active Learning Literature Survey [Internet] . 2009 [cited 2023 Aug 26]. Available from: http://digital.library.wisc.edu/1793/60660 18. ↵ Wallace BC , Small K , Brodley CE , Trikalinos TA . Active Learning for Biomedical Citation Screening . In: Proceedings of the 16th ACM SIGKDD International Conference on Knowledge Discovery and Data Mining [Internet] . New York, NY, USA : Association for Computing Machinery ; 2010 . p. 173 – 182 . (KDD ’10). Available from: https://doi-org.libaccess.lib.mcmaster.ca/10.1145/1835804.1835829 19. ↵ Longpre S , Reisler J , Huang EG , Lu Y , Frank A , Ramesh N , et al. Active Learning Over Multiple Domains in Natural Language Tasks . 2022 Feb 1; Available from: http://arxiv.org/abs/2202.00254 20. ↵ van de Schoot R , de Bruin J , Schram R , Zahedi P , de Boer J , Weijdema F , et al. An open source machine learning framework for efficient and transparent systematic reviews . Nat Mach Intell . 2021 Feb 1; 3 ( 2 ): 125 – 33 . OpenUrl 21. Wallace BC , Small K , Brodley CE , Lau J , Trikalinos TA . Modeling annotation time to reduce workload in comparative effectiveness reviews. IHI’10 - Proceedings of the 1st ACM International Health Informatics Symposium . 2010 ; 28 – 35 . 22. ↵ Boetje J , van de Schoot R . The SAFE procedure: a practical stopping heuristic for active learning-based screening in systematic reviews and meta-analyses . Syst Rev [Internet] . 2024 Mar 1; 13 ( 1 ): 81 . Available from: https://systematicreviewsjournal.biomedcentral.com/articles/10.1186/s13643-024-02502-7 OpenUrl 23. ↵ Williamson PR , Altman DG , Bagley H , Barnes KL , Blazeby JM , Brookes ST , et al. The COMET Handbook: version 1.0 . Trials [Internet] . 2017 ; 18 ( 3 ): 280 . Available from : doi: 10.1186/s13063-017-1978-4 OpenUrl CrossRef PubMed 24. Kirkham JJ , Gorst S , Altman DG , Blazeby JM , Clarke M , Tunis S , et al. Core Outcome Set-STAndardised Protocol Items: the COS-STAP Statement . Trials [Internet ]. 2019 ; 20 ( 1 ): 116 . Available from : doi: 10.1186/s13063-019-3230-x OpenUrl CrossRef PubMed 25. ↵ Kearney A , Harman NL , Rosala-Hallas A , Beecher C , Blazeby JM , Bower P , et al. Development of an online resource for recruitment research in clinical trials to organise and map current literature . Clinical Trials . 2018 Dec 31; 15 ( 6 ): 533 – 42 . OpenUrl PubMed 26. ↵ Kusa W , Lipani A , Knoth P , Hanbury A . An analysis of work saved over sampling in the evaluation of automated citation screening in systematic literature reviews . Intelligent Systems with Applications . 2023 May; 18 : 200193 . OpenUrl 27. ↵ Wallace BC , Trikalinos TA , Lau J , Brodley C , Schmid CH . Semi-automated screening of biomedical citations for systematic reviews . BMC Bioinformatics . 2010 Dec 26; 11 ( 1 ): 55 . OpenUrl CrossRef PubMed 28. ↵ Holland J , Haynes RB , McMaster PLUS Team Health Information Research Unit . McMaster Premium Literature Service (PLUS): an evidence-based medicine information service delivered on the Web. AMIA Annu Symp Proc [Internet ]. 2005 ; 340 – 4 . Available from: http://www.ncbi.nlm.nih.gov/pubmed/16779058 29. ↵ Haynes RB , Holland J , Cotoi C , McKinlay RJ , Wilczynski NL , Walters LA , et al. McMaster PLUS: A Cluster Randomized Clinical Trial of an Intervention to Accelerate Clinical Use of Evidence-based Information from Digital Libraries . Journal of the American Medical Informatics Association . 2006 Nov 1; 13 ( 6 ): 593 – 600 . OpenUrl CrossRef PubMed 30. ↵ Bailly A , Blanc C , Francis É , Guillotin T , Jamal F , Wakim B , et al. Effects of dataset size and interactions on the prediction performance of logistic regression and deep learning models . Comput Methods Programs Biomed . 2022 Jan; 213 : 106504 . OpenUrl CrossRef PubMed 31. ↵ Ferdinands G , Schram R , de Bruin J , Bagheri A , Oberski DL , Tummers L , et al. Performance of active learning models for screening prioritization in systematic reviews: a simulation study into the Average Time to Discover relevant records . Syst Rev . 2023 Jun 20; 12 ( 1 ): 100 . OpenUrl CrossRef PubMed 32. ↵ O’Mara-Eves A , Thomas J , McNaught J , Miwa M , Ananiadou S . Using text mining for study identification in systematic reviews: A systematic review of current approaches . Syst Rev [Internet] . 2015 ; 4 (( Miwa ) Toyota Technological Institute, 2-12-1 Hisakata, Tempaku-ku, Nagoya 468-8511 , Japan ): 5 . Available from: http://ovidsp.ovid.com/ovidweb.cgi?T=JS&PAGE=reference&D=emed16&NEWS=N&AN=605593033 OpenUrl 33. ↵ Ilhan HO , Amasyalı MF . Active Learning as a Way of Increasing Accuracy . International Journal of Computer Theory and Engineering . 2014 Dec; 6 ( 6 ): 460 – 5 . OpenUrl 34. ↵ Teijema JJ , de Bruin J , Bagheri A , van de Schoot R . Large-scale simulation study of active learning models for systematic reviews . Int J Data Sci Anal . 2025 May 2; View the discussion thread. Back to top Previous Next Posted October 15, 2025. Download PDF Data/Code Email Thank you for your interest in spreading the word about medRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following A machine learning model to support the screening for methods guidance articles in MEDLINE: A performance evaluation of ASReview simulation mode Message Subject (Your Name) has forwarded a page to you from medRxiv Message Body (Your Name) thought you would like to see this page from the medRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share A machine learning model to support the screening for methods guidance articles in MEDLINE: A performance evaluation of ASReview simulation mode Wael Abdelkader , Daniel Xie , Cynthia Lokker , Lingyang Chu , Stefan Schandelmaier , Ashirbani Saha , Muhammad Afzal , Alfonso Iorio medRxiv 2025.10.13.25337935; doi: https://doi.org/10.1101/2025.10.13.25337935 Share This Article: Copy Citation Tools A machine learning model to support the screening for methods guidance articles in MEDLINE: A performance evaluation of ASReview simulation mode Wael Abdelkader , Daniel Xie , Cynthia Lokker , Lingyang Chu , Stefan Schandelmaier , Ashirbani Saha , Muhammad Afzal , Alfonso Iorio medRxiv 2025.10.13.25337935; doi: https://doi.org/10.1101/2025.10.13.25337935 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Health Informatics Subject Areas All Articles Addiction Medicine (567) Allergy and Immunology (863) Anesthesia (297) Cardiovascular Medicine (4411) Dentistry and Oral Medicine (443) Dermatology (380) Emergency Medicine (606) Endocrinology (including Diabetes Mellitus and Metabolic Disease) (1505) Epidemiology (15205) Forensic Medicine (30) Gastroenterology (1119) Genetic and Genomic Medicine (6574) Geriatric Medicine (666) Health Economics (994) Health Informatics (4511) Health Policy (1365) Health Systems and Quality Improvement (1608) Hematology (537) HIV/AIDS (1263) Infectious Diseases (except HIV/AIDS) (15903) Intensive Care and Critical Care Medicine (1103) Medical Education (620) Medical Ethics (144) Nephrology (665) Neurology (6573) Nursing (345) Nutrition (998) Obstetrics and Gynecology (1139) Occupational and Environmental Health (954) Oncology (3319) Ophthalmology (967) Orthopedics (369) Otolaryngology (420) Pain Medicine (435) Palliative Medicine (129) Pathology (662) Pediatrics (1689) Pharmacology and Therapeutics (691) Primary Care Research (710) Psychiatry and Clinical Psychology (5421) Public and Global Health (9205) Radiology and Imaging (2191) Rehabilitation Medicine and Physical Therapy (1367) Respiratory Medicine (1191) Rheumatology (593) Sexual and Reproductive Health (709) Sports Medicine (529) Surgery (709) Toxicology (99) Transplantation (288) Urology (265) (function(){function c(){var b=a.contentDocument||a.contentWindow.document;if(b){var d=b.createElement('script');d.innerHTML="window.__CF$cv$params={r:'9fe8f95d4925ad07',t:'MTc3OTI1NTI4NQ=='};var a=document.createElement('script');a.src='/cdn-cgi/challenge-platform/scripts/jsd/main.js';document.getElementsByTagName('head')[0].appendChild(a);";b.getElementsByTagName('head')[0].appendChild(d)}}if(document.body){var a=document.createElement('iframe');a.height=1;a.width=1;a.style.position='absolute';a.style.top=0;a.style.left=0;a.style.border='none';a.style.visibility='hidden';document.body.appendChild(a);if('loading'!==document.readyState)c();else if(window.addEventListener)document.addEventListener('DOMContentLoaded',c);else{var e=document.onreadystatechange||function(){};document.onreadystatechange=function(b){e(b);'loading'!==document.readyState&&(document.onreadystatechange=e,c())}}}})();
Text is read by the "Ask this paper" AI Q&A widget below.
Extraction quality varies by source — PMC NXML preserves structure
cleanly, OA-HTML may include some navigation residue, and OA-PDF can
have broken hyphenation. The publisher copy
(via DOI)
is the canonical version.