Full text
53,002 characters
· extracted from
preprint-html
· click to expand
T-Rx: A toolbox for reproducible processing of prescriptions (Rx) from electronic health records | medRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-P4HH5NV'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search T-Rx: A toolbox for reproducible processing of prescriptions (Rx) from electronic health records View ORCID Profile Chris Wai Hang Lo , View ORCID Profile Dale Handley , View ORCID Profile Oliver Pain , View ORCID Profile Michelle Kamp , View ORCID Profile Alexandra C. Gillett , View ORCID Profile Matthew H. Iveson , View ORCID Profile Chiara Fabbri , View ORCID Profile Katherine G. Young , AMBER Research Team , View ORCID Profile Cathryn M. Lewis doi: https://doi.org/10.1101/2025.10.07.25336002 Chris Wai Hang Lo 1 Social, Genetic & Developmental Psychiatry Centre, Institute of Psychiatry, Psychology and Neuroscience, King’s College London , London, United Kingdom Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Chris Wai Hang Lo Dale Handley 1 Social, Genetic & Developmental Psychiatry Centre, Institute of Psychiatry, Psychology and Neuroscience, King’s College London , London, United Kingdom Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Dale Handley Oliver Pain 2 Department of Basic and Clinical Neuroscience, Institute of Psychiatry, Psychology and Neuroscience, King’s College London , London, United Kingdom Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Oliver Pain Michelle Kamp 1 Social, Genetic & Developmental Psychiatry Centre, Institute of Psychiatry, Psychology and Neuroscience, King’s College London , London, United Kingdom Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Michelle Kamp Alexandra C. Gillett 1 Social, Genetic & Developmental Psychiatry Centre, Institute of Psychiatry, Psychology and Neuroscience, King’s College London , London, United Kingdom 3 National Institute for Health Research Maudsley Biomedical Research Centre at South London and Maudsley NHS Foundation Trust and King’s College London , London, United Kingdom Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Alexandra C. Gillett Matthew H. Iveson 4 School of Neurological and Cardiovascular Sciences, University of Edinburgh , Edinburgh, United Kingdom Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Matthew H. Iveson Chiara Fabbri 5 Department of Biomedical and Neuromotor Sciences, University of Bologna , Bologna, Italy Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Chiara Fabbri Katherine G. Young 6 Department of Clinical and Biomedical Sciences, University of Exeter Medical School , Exeter, United Kingdom Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Katherine G. Young Cathryn M. Lewis 1 Social, Genetic & Developmental Psychiatry Centre, Institute of Psychiatry, Psychology and Neuroscience, King’s College London , London, United Kingdom 3 National Institute for Health Research Maudsley Biomedical Research Centre at South London and Maudsley NHS Foundation Trust and King’s College London , London, United Kingdom 7 Department of Medical & Molecular Genetics, King’s College London , London, United Kingdom Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Cathryn M. Lewis For correspondence: cathryn.lewis{at}kcl.ac.uk Abstract Full Text Info/History Metrics Supplementary material Data/Code Preview PDF Abstract Linkage between population-wide biobanks and electronic health records (EHRs) opens new opportunities to study the genetic and epidemiological underpinnings of treatment outcomes, a key step forward in delivering precision medicine. However, challenges include complexities in data extraction for longitudinal analyses, and the absence of reproducible phenotyping algorithms. Here, we present T-Rx, an open-source R package to streamline the processing of prescription and dispensing records, to enable reproducible and scalable analysis across EHR databases. T-Rx consists of three modules to derive treatment-related phenotypes from uncleaned prescription records: (1) Extraction and Imputation Module for extracting and imputing prescription details; (2) Exposure Ascertainment Module for converting prescriptions to longitudinal exposure periods; and (3) Phenotyping Module for creating reproducible proxy phenotypes that capture treatment and response patterns. We tested the utility of T-Rx in UK Biobank primary care records, with strength and quantity information extracted and imputed for 2,721,921 antidepressant and 430,705 antipsychotic prescriptions. The extraction functions were validated using oral hypoglycemic agent prescriptions in Clinical Practice Research Datalink Aurum, showing comparable performances to extraction using the NHS Dictionary of Medicines and Devices (dm+d) codes. The Exposure Ascertainment Module of T-Rx converts discrete prescription or dispensing events into longitudinal exposure periods in one-line R commands, with customizable parameters to account for real-world treatment complexities. The Phenotyping Module takes prescriptions as direct user input and returns analysis -ready data frames. Current phenotyping algorithms include antidepressant switching and treatment-resistant depression. The phenotyping functions also allow flexible parameter choices, such as treatment episode windows, definitions for switching and quality control criteria. Researchers can contribute phenotyping algorithms to T-Rx for reproducible use. T-Rx improves the accessibility of prescription information in biobanks, and analysis of dosage- and treatment-patterns across therapeutic areas. T-Rx contributes to open science through harmonized phenotypic definitions and reproducible analyses of proxy treatment outcomes. Introduction Since the computerization of healthcare records and electronic prescribing, healthcare systems worldwide have established extensive electronic health record (EHR) database initiatives [ 1 , 2 ]. EHR databases, characterized by large sample sizes and a broad range of phenotypic variables collected from routine clinical practice [ 3 ], have transformed medical research by enabling widescale epidemiological studies from clinical records. More recently, data linkage has been established between EHRs and biobanks with genetic data, for example, UK Biobank (UKB) and iPSYCH [ 4 – 6 ]. This rich combination of phenotypic and genetic data opens new opportunities to study individual differences in disease and treatment trajectories longitudinally, a key step forward in precision medicine [ 7 ]. Open-source platforms like the Health Data Research UK Phenotype Library [ 8 ] host reproducible phenotyping algorithms for diseases and conditions, but we lack a similar repository for treatment-related phenotypes. Longitudinal analyses of EHR data rely on both structured data (e.g., fields documenting information on date of birth, ethnicity) and unstructured data (e.g., free-text narratives describing symptoms or side effects) collected in routine clinical practice. Prescriptions in many EHR databases come in semi-structured format and contain information on medication classes and prescription dates [ 9 ]. Despite their value, semi-structured records pose challenges for widespread use in the research community. Prescription record details of dosage, quantity, and frequency of administration are not structured in readily usable formats [ 9 , 10 ], if available at all. These prescribing events are usually discrete and cross-sectional in nature, thus they are not readily available for longitudinal analysis of treatment patterns. These caveats limit the ability of researchers to infer treatment episodes (i.e., how long the patient is exposed to a medication longitudinally) [ 11 ], and requires manual extraction to model drug exposure [ 12 – 14 ], which is difficult to scale and limits reproducibility across databases [ 9 ]. Clinical trials and research studies collect detailed information on symptom changes and other outcomes useful in assessing treatment effects, but EHRs often lack such longitudinal measurements. Therefore, proxy outcomes from medical records are necessary to capture disease characteristics and treatment trajectories [ 15 ]. Treatment-resistant depression (TRD) for instance, can be defined from EHRs as at least two antidepressant switches after adequate duration of exposure [ 12 , 16 ]. This working definition has been useful in driving new, large-scale genetic [ 12 , 17 ] and clinical [ 18 , 19 ] studies of antidepressant non-response, overcoming the sample size limitations of clinical trials. However, the phenotyping algorithms for treatment outcomes are not harmonized across studies [ 15 , 20 ], making cross-validation of results challenging. In turn, this limits the scalability and integration of analyses across EHR databases [ 21 , 22 ]. Harmonized algorithms for treatment phenotypes would make analyses reproducible and accessible to researchers with limited experience working with EHRs. These tools would enhance the value of EHRs and linked biobanks for research on treatment patterns and response [ 21 ]. Recognizing these challenges, we introduce T-Rx, a toolbox for transferrable, reproducible extraction of prescriptions (Rx) from EHRs. This comprehensive R package provides a simplified approach for extracting and imputing prescription details using regular expression (regex) patterns, and includes a library of phenotyping algorithms enabling researchers to quickly and reproducibly scale up analyses. By simplifying data wrangling and phenotyping, T-Rx serves as an open-source tool to enhance the accessibility of complex EHR data in diverse research and clinical contexts. Here we detail the features, implementation, and application of the T-Rx package, demonstrating its utility across EHR databases and disease areas, including psychiatric disorders and cardiometabolic diseases. Overview of the T-Rx package T-Rx is structured as three modules: Extraction and Imputation Module : provides algorithms to extract and impute prescription details using regex patterns relevant in EHR databases; Exposure Ascertainment Module: provides algorithms to infer or construct treatment episodes longitudinally from prescriptions; Phenotyping Module: hosts a library of flexible phenotyping algorithms used in published literature. T-Rx is hosted in R environment [ 23 ], which allows researchers to easily install, scale and reproduce EHR data analyses. Figure 1A gives an overview of T-Rx functionalities. Download figure Open in new tab Figure 1. (A) Overview of T-Rx modules; (B) Extraction and Imputation Module of T-Rx, with UK Biobank primary care prescription records as an example; (C) Imputation of strengths and quantities with T-Rx; (D) Exposure Ascertainment Module of T-Rx; (E) Phenotyping Module of T-Rx, using antidepressant switching and treatment resistant depression as example functions, showing example prescriptions as input and phenotype data frames as outputs Legends Created in BioRender. Lo, C. (2025) https://BioRender.com/2izbwra . Figures 1B/C demonstrate the processes of extracting and imputing product strengths from prescription strings, using UK Biobank primary care records as example. This can also be applied to other electronic health records, where dosage (i.e.: the actual amount of medication taken) instead of product strengths are available. Abbreviations QC = quality control; Rx = prescription. Extraction and Imputation Module Overview The Extraction and Imputation Module of T-Rx extracts details from raw prescription records using regex patterns. The current extraction functions are targeted for United Kingdom prescribing databases, with prescribing data formats from NHS England [ 24 ], such as UKB [ 6 ], Clinical Practice Research Datalink (CPRD) [ 25 ] and the Genes and Health Study [ 26 ]. The imputation functions infer missing details, including medication strength and quantity, from the prescription data fed into the function, and performs processing on multi-ingredient products ( Figure 1B ). Extraction of prescription records The extraction functions rely on regex pattern algorithms to extract strengths and quantities of medications in prescription records, using strength units (e.g., mg , mg/ml ) and dosage forms (e.g., tab , suspension ) as tags for extraction. This method is more flexible to strings describing pharmaceutical products, and therefore not limited to specific coding systems in prescriptions. In T-Rx, these strings of tags (see Supplementary Materials for codes) are specified by users based on prior knowledge of the medication(s) of interest and differ across therapeutic areas. T-Rx functions then extract the numbers that precede the user-inputted “tags” as strengths and quantities ( Figure 1B ). The three T-Rx functions listed below extract strengths, quantities and multipliers of medications from the raw prescription records, as depicted in Figure 1B : strength_extract() : extract strengths (highlighted in yellow) as numbers that precede the user-input strength units, e.g., mg (in green); quantity_extract(): extract quantities, e.g., number of tablets (in pink) as numbers that precede the user-input dosage forms, e.g., tablets (in blue); and multiplier_extract(): extract multiplier information whenever necessary, e.g., number of packs (in grey), after running the functions above. Imputation of prescription records The imputation functions infer strength and quantity information from the outputs of T-Rx extraction functions or as directly inputted by users. From this data frame, an imputation reference panel of strengths and quantities is created. The most commonly occurring strength or quantity would be used to impute and replace missing values (i.e., mode imputation). Using user-provided prescription information has the merit of capturing prescription patterns from relevant data, which could differ across databases and regions of practice. However, this mode of imputation might work less well in datasets where details of prescription were systematically absent. A logic flowchart of imputation functions is described in Figure 1C . The two imputation functions are: ‘ strength_impute() ’: performs imputation on strengths of products, and expands multi-strength products into multiple rows with strengths of each active ingredient mapped (highlighted in blue in Figure 1C ); ‘ quantity_impute() ’: performs imputation on quantities of products. Additional functionalities for inferring prescription information In datasets such as UKB, EHR linkage was established with multiple providers [ 6 ], making the format of prescriptions particularly heterogeneous, and further inference is required to extract the strength and quantity information. These two additional functions ‘ multi_num_infer() ’ and ‘ duration_handling() ’ deal with the complexity of UKB primary care prescription records, such as inferring quantities from prescriptions without dosage form tags , or with strings representing duration of prescriptions being present (e.g., 4/52 as 4 weeks , 3/12 as 3 months ). The details of additional functionalities are described in Supplementary Materials . Exposure Ascertainment Module – Overview Prescription data is usually provided as a single row for each event for every individual, which needs to be merged to form a treatment episode ( Figure 1D ). One common approach is to construct treatment episodes using continuous prescription records – two prescriptions can be assigned to the same treatment episode if the periods of coverage overlap ( Figure 2A ). However, accurately modelling periods of medication exposure is not straightforward. A window between prescribing periods should be permitted to account for real-world treatment complexity, such as nudges in follow-up periods due to medication stockpiling, so that misclassification as treatment discontinuation can be avoided [ 27 ]. In addition, the duration of exposure is often unavailable, and assumptions need to be made to infer treatment episodes ( Figure 2B ), such as using daily defined doses and prescription quantities [ 10 ]. Download figure Open in new tab Figure 2. Exposure ascertainment module functionality for a three-prescription example: (A) sertraline prescriptions (with start and end dates of prescriptions available); (B) sertraline prescriptions (without end dates of prescriptions); (C) sertraline prescriptions (as table); (D) treatment episodes when sertraline prescriptions were merged into prescribing episodes, without gaps allowed between prescriptions; (E) treatment episodes when sertraline prescriptions were merged into prescribing episodes, with a 7-day gap allowed between prescriptions; (F), (G) treatment episodes as tables from (D) and (E), as outputs from rx_merge() . Legends Details of functions are available on the T-Rx website at: https://chrislowh.github.io/T-Rx/ . T-Rx provides two different algorithms to construct treatment episodes based on EHR data availability: ‘ rx_merge() ’: when coverage of each prescription is available, aggregate them into the same episode if the coverages overlap, with a customizable parameter to denote time gaps permitted ( Figure 2D & 2E ); ‘ rx_infer() ’: when coverage of each prescription is not available, infer treatment duration and episodes by “repeated prescriptions” ( Figure 2B ). The functions require prescription data containing drug names and dates as input from users and return treatment episodes suitable for further analysis. Details of the functions are available in Supplementary Materials. Phenotyping Module – Overview Information on response to treatment or reasons for changing (or stopping) a prescribed drug is rarely given in structured EHR data. Prescription data extracted by T-Rx modules can be used to create proxy or surrogate measures, including length of prescribing episodes, drug switching or treatment resistance. The Phenotyping Module of T-Rx hosts a library of phenotyping algorithms for creating proxy treatment outcomes, using definitions from published literature [ 12 , 14 ]. Phenotyping processes are easy to run, being structured into R functions with optional parameters. Analyses of T-Rx datasets can then be applied across EHRs and biobanks, giving consistent phenotyping processes. To contribute to open science, researchers can submit their phenotyping algorithms, enabling others to apply the same algorithm reproducibly. This module has the potential for further expansion into a repository of proxy-phenotyping related to treatment, similar to the Health Data Research Phenotype Library [ 8 ]. Figure 1E shows current phenotyping functions available in T-Rx, with user inputs and expected outputs. Application Extraction and Imputation Module Demonstration – UKB As an illustration, we apply the T-Rx Extraction and Imputation Module to antidepressant, antipsychotic and lithium primary care prescriptions in UKB [ 6 , 12 ]. In the UKB, linkage to prescriptions in primary care were established for ∼230,000 participants, recorded under READ v2, British National Formulary (BNF) and the NHS Dictionary of Medicines and Devices (dm+d) codes. Details on the UKB sample and primary care prescriptions were described in Supplementary Methods . The READv2, BNF and dm+d codes used to extract antipsychotic prescriptions are summarized in Supplementary table 1 , and those used to extract antidepressant prescriptions are described elsewhere [ 12 ]. As illustration, samples of UKB prescriptions are shown in Supplementary figure 2. UKB prescriptions in Wales were removed due to the absence of strength and quantity information. After removal, 2,718,545 antidepressant prescriptions were available for 82,647 UKB participants, and 430,705 antipsychotics and lithium prescriptions were available for 38,645 UKB participants. Supplementary table 2 and Supplementary figure 4 describe the summary measures by each drug available in UKB primary care records. To extract product strengths from prescriptions, lists of strings specifying strength units for both solid (e.g., tablets, capsules ) and liquid (e.g., injections, suspensions ) dosage forms are given as arguments ( liquid_strength_unit, solid_strength_unit) in the ‘ strength_extract() ’ R function. An illustration of R codes is provided in Supplementary methods . After running ‘ strength_extract() ’, the strengths of 2,320,226 (85.3%) antidepressant and 349,705 (81.2%) antipsychotic and lithium prescriptions were extracted. The strengths of some prescriptions cannot be extracted due to strength and strength unit values being absent from the antidepressant, antipsychotic or lithium prescriptions. For prescriptions without strengths extracted, the strengths were imputed using ‘ strength_impute() ’ function ( Figure 1C ). Multi-ingredient products were also handled and expanded to single strength products. This provides strength values for 99% of antidepressant prescriptions (2,721,931/2,721,987 prescriptions, after expansion of multi-ingredient products, Supplementary figure 5 ) and all antipsychotic prescriptions (N = 430,705) in UKB. The strength distributions of products after running T-Rx are summarized in Figure 3 , and Supplementary tables 3 and 4 . Download figure Open in new tab Figure 3. Distribution of strengths of (A) antidepressant and (B) antipsychotic prescriptions in UK Biobank primary care records after strength information extracted or imputed by T-Rx. Column labels show strengths and proportions of prescriptions at each strength level. Legends The strength distributions of SSRIs, SNRIs, TCAs, TeCAs and mirtazapine are shown for antidepressants, while those for FGA, SGA and lithium are shown for antipsychotic prescriptions. Only products with more than 500 prescriptions in UKB primary care records were included, with labels only for strengths of products accounting for more than 5% of all products with the same drug name. Details of strength distributions are summarized in Supplementary tables 3 and 4. Abbreviations FGA = first-generation antipsychotics; SGA = second-generation antipsychotics; SNRI = serotonin-norepinephrine reuptake inhibitors; SSRI = selective serotonin reuptake inhibitors; TCA = tricyclic antidepressants; TeCA = tetracyclic antidepressants; UKB = UK Biobank. Validation – Clinical Research Practice Datalink (CPRD) Aurum We tested the validity of T-Rx for extracting strengths and strength units using prescriptions of oral hypoglycemic agents (OHAs) [ 28 ] in individuals with Type 2 diabetes (T2D) in the CPRD Aurum dataset [ 25 , 29 , 30 ]. CPRD Aurum is a UK primary care database containing anonymized electronic health records from contributing general practices. OHA prescription records, including quantity, were extracted using CPRD Aurum unique product codes. These link to a product dictionary containing dm+d derived information, including product name and strength, where available ( Supplementary table 5 ). For the purposes of validation, we randomly selected 100,000 individuals with T2D and at least one OHA prescription event, resulting in 8,707,546 prescriptions. We applied the ‘ strength_extract() ’ function to product name data (e.g., Metformin 500mg tablets ) and assessed concordance with strengths obtained from linked dm+d codes. A detailed description of the validation sample is given in Supplementary Materials , with information on the number of CPRD participants and prescriptions in Supplementary table 7 . Of the 8,707,546 prescriptions in the validation dataset, dm+d product strength information was available for 8,707,513 prescriptions, all of which can also be extracted with T-Rx ( Figure 4 ). The strength and strength unit from T-Rx agreed with dm+d values for 98.6% of prescriptions (N = 8,586,964). The small number of discrepancies between T-Rx and dm+d outputs are due to differences in rounding for concentrations of liquid dosage formulations (46,076; 0.5%), or differences in string formatting in products with multiple strengths (74,473; 0.9%). As the presence of multi-strength products made comparison across methods difficult, we have provided sample multi-strength product codes in the validation dataset in Supplementary table 8 , showing the outputs of strength extractions from CPRD Aurum and T-Rx for comparison. Download figure Open in new tab Figure 4. Sankey diagram to compare the performance of strength extraction functions in T-Rx with oral hypoglycemic agent prescriptions in CPRD Legends Figures expressed as number of prescriptions (percentage). Exposure Ascertainment Module The ‘ rx_merge() ’ function allows users to merge prescriptions from multiple data sources, taking into account the differences in data availability of prescriptions. Prescriptions can be merged based on prescription dates, where prescriptions were merged if the end date of the previous prescription overlapped with the start date of the next prescription ( Figure 2A ). This would normally be the preferred approach if the start and end dates of prescriptions are available in the dataset. After merging, the function returns a data frame with prescribing episodes for all individuals ( Figure 2F and 2G ). Prescription episodes may differ, depending on the gaps allowed between prescriptions, and data availability, which can be adjusted using arguments in the R function. Phenotyping Module – example functions In the T-Rx phenotyping module, users can create proxy phenotypes with a one-line R command, by specifying the data frame of prescription records or episodes (with information on column names) and quality control parameters. These parameters allow users to adjust the specificity and sensitivity of phenotyping algorithms. Here, we demonstrate two example functions to identify antidepressant switching [ 14 ] and treatment-resistant depression events [ 12 ], based on open-source codes in the published literature. Antidepressant switching Switching is a proxy phenotype to capture medication non-response using prescription records, with phenotyping details described elsewhere [ 14 , 31 ]. The phenotyping scripts have three steps to identify switching events, based on gaps between prescription dates of two different medications, imposing additional quality control criteria and identifying comparable controls [ 14 ]. To run the phenotyping function ‘ switch_Lo2025() ’ ( Figure 1E ), users are required to specify the prescription data frame, the column names for drug and prescription dates, windows to capture switching events and quality control parameters. Figure 5 shows a schematic diagram detailing the process from a user perspective, including the prescription data frame required, and a visualization of prescribing journeys for four sample patients. Running ‘ switch_Lo2025() ’ returns a data frame of switchers (as wide format, Figure 5C ), containing drug information before and after switching, time to switch and index date, i.e., the first prescription date of the pre-switch drug. Download figure Open in new tab Figure 5. Sample use of T-Rx phenotyping functions, with (A) prescription data frame as user input, (B) visualization of prescriptions from input, (C) output phenotype data frame with drug switching (switch_Lo2025()) and (D) output phenotype data frame with treatment resistant depression (TRD_Fabbri2021()). Legends Details of functions creating the two phenotypes are at: https://chrislowh.github.io/T-Rx/ . Treatment-resistant depression TRD was defined and validated previously using at least two antidepressant switches within 14 weeks, with further quality control parameters [ 12 ]. T-Rx concatenated the phenotyping algorithm for ease of use as ‘ TRD_Fabbri2021() ’ ( Figure 1E ). This allows ready comparison with other TRD algorithms, given the heterogeneity of definitions in the literature [ 32 ]. Running the function returns a data frame containing TRD patients, with details on antidepressant prescriptions received ( Figure 5D ). Discussion T-Rx serves as a tool to improve the accessibility of prescription records for researchers, by simplifying data cleaning and phenotyping processes related to treatment. T-Rx is particularly useful in large-scale clinical epidemiology studies, where data extraction steps are often performed manually [ 12 , 13 ], requiring extensive knowledge of EHRs and treatment patterns. T-Rx automates these processes in a user-friendly R framework, allowing cleaned prescription datasets to be easily created and widely used by the research community. In particular, T-Rx makes prescription dosages more accessible in population-wide biobanks, facilitating pharmacokinetic [ 33 ] and pharmacogenomic studies [ 34 , 35 ] for personalizing medication dosing. Defining clinically relevant proxy outcomes that capture treatment or response characteristics of medications from prescriptions is challenging [ 22 ]. Constructing these outcomes requires complex coding, as prescriptions must be assessed longitudinally rather than as discrete events. In addition, quality control parameters are required to ensure the clinical relevance of the derived proxy outcomes [ 12 , 14 ]. The heterogeneity in data availability across datasets brings further issues when using prescription-derived phenotypes as a study outcome in meta-analyses. Applying consistent algorithms to define treatment outcomes facilitates cross -study comparisons, reducing barriers to reproducible analyses. Comparison to current tools To our knowledge, T-Rx is the most comprehensive tool to streamline the wrangling processes required to convert raw prescriptions to readily analyzed phenotypes. Most existing tools focus on a single purpose, such as estimating drug exposure duration [ 36 ] or visualizing prescription data [ 37 ]. The types of workflows in T-Rx are typically undertaken in data wrangling processes in pharmaco-epidemiological studies, but codes and algorithms are seldom shared on open-source platforms, so other researchers cannot reproduce the dataset for further analyses. We hope T-Rx will provide a user-friendly, transparent and reproducible framework for prescription cleaning and phenotype creation. This is particularly relevant for the Phenotyping Module, which will support important initiatives to harmonize treatment definitions in real-world data [ 22 ]. Hosting published algorithms as open-source codes will facilitate the validation of treatment definitions across studies in a reproducible fashion. The Extraction and Imputation Module in T-Rx utilizes regex patterns for the extraction of prescription details, in contrast to some of the workflows under development with natural language processing (NLP) [ 38 , 39 ]. These workflows primarily focus on named-entity recognition tasks and show excellent performance by mapping drug names to active pharmaceutical ingredients, but are less satisfactory for extracting prescribed dosages [ 38 ]. NLP algorithms are also computationally intensive, which could be a barrier in resource-limited settings [ 40 ]. Rule-based algorithms for extracting prescription details in T-Rx can also serve as a performance benchmark for future development of NLP algorithms. Limitations and Future Directions The use of T-Rx might be limited by heterogeneity in the formats of prescription records across healthcare systems. Depending on data availability in EHR databases, these rule-based algorithms for extracting prescription details may not always be necessary. The aim of T-Rx is to offer solutions for prescription processing that are adaptable to different EHR systems, such that researchers can decide which modules to run to prepare the prescription or dispensing data for analysis. In databases where the extraction of prescription details is necessary, we have demonstrated satisfactory performance using prescription records in UKB and CPRD across different therapeutic areas. The imputation module can also handle missingness in prescription details from any EHR prescribing database [ 41 , 42 ]. The episode ascertainment module of T-Rx is generalizable to most EHR databases, where prescriptions are typically available as discrete events and require conversion to prescribing episodes for longitudinal analysis. The overall utility of T-Rx in the wider community of EHR analysis requires further assessment, which can be addressed in partnership with researchers accessing data from other prescription databases. Phenotyping algorithms in T-Rx are currently limited to proxy phenotypes related to antidepressants. The phenotype library in T-Rx can be expanded with contributions of novel phenotyping algorithms for inclusion in T-Rx. Conclusion T-Rx facilitates pharmaco-epidemiological studies using EHRs by improving the accessibility of prescriptions, and by simplifying data extraction and phenotyping processes to create reproducible phenotypes. This is an important step in precision medicine, enabling large-scale studies to be conducted to yield more valuable insights. T-Rx contributes to open science by enabling complex phenotyping algorithms to be hosted in R, allowing the research community to re-create phenotypes in a reproducible fashion across systems. Data availability Tutorials and instructions for T-Rx are provided at https://chrislowh.github.io/T-Rx/ . The UK Biobank data was accessed via project 82087. Data access to UKB biobank is available for approved researchers at https://www.ukbiobank.ac.uk/enable-your-research . This project was conducted under CPRD study reference ID 23_002544. Researchers wishing to access CPRD data must obtain the necessary approvals and data access agreements. Data Availability Tutorials and instructions for T-Rx are provided at https://chrislowh.github.io/T-Rx/ . The UK Biobank data was accessed via project 82087. Data access to UKB biobank is available for approved researchers at https://www.ukbiobank.ac.uk/enable-your-research . This project was conducted under CPRD study reference ID 23_002544. Researchers wishing to access CPRD data must obtain the necessary approvals and data access agreements. https://chrislowh.github.io/T-Rx/ Funding This research was funded by Wellcome Mental Health Award (226770/Z/22/Z) and part-funded by MRC project grant (MR/X009815/1) and the National Institute for Health and Care Research (NIHR) Maudsley Biomedical Research Centre (BRC) at South London and Maudsley NHS Foundation Trust and King’s College London. The views expressed are those of the author(s) and not necessarily those of the NIHR or the Department of Health and Social Care. Ethical approval for observational studies using the Clinical Practice Research Datalink is granted by the NHS Health Research Authority (Derby Research Ethics Committee; REC reference 21/EM/0265). Individual patient consent is not required. OP is supported by a Sir Henry Wellcome Postdoctoral Fellowship [222811/Z/21/Z]. MHI is supported by the HDR UK DATAMIND hub, which is funded by the UK Research and Innovation grant MR/W014386/1, by the Wellcome Trust (220857/Z/20/Z; 104036/Z/14/Z; 216767/Z/19/Z) and by a Research Data Scotland Accelerator Award (RAS-24-2). Competing Interests CML sits on the Myriad Neuroscience Scientific Advisory Board. CML and OP provide consultancy services for UCB Pharma. The remaining authors declare that there are no competing interests. Acknowledgements For the purposes of open access, the author has applied a Creative Commons Attribution (CC BY) licence to any Accepted Author Manuscript version arising from this submission. We extend our gratitude to the contributions of other investigators on the AMBER: Antidepressant Medications: Biology, Exposure & Response project, with the list of investigators provided in Supplementary Materials . References ↵ Benson T . Why general practitioners use computers and hospital doctors do not--Part 1: incentives . BMJ . 2002 ; 325 : 1086 – 9 . doi: 10.1136/bmj.325.7372.1086 OpenUrl FREE Full Text ↵ Casey JA , Schwartz BS , Stewart WF , et al. Using Electronic Health Records for Population Health Research: A Review of Methods and Applications . Annu Rev Public Health . 2016 ; 37 : 61 – 81 . doi: 10.1146/annurev-publhealth-032315-021353 OpenUrl CrossRef PubMed ↵ Wolford BN , Willer CJ , Surakka I . Electronic health records: the next wave of complex disease genetics . Hum Mol Genet . 2018 ; 27 : R14 – 21 . doi: 10.1093/hmg/ddy081 OpenUrl CrossRef PubMed ↵ Bycroft C , Freeman C , Petkova D , et al. The UK Biobank resource with deep phenotyping and genomic data . Nature . 2018 ; 562 : 203 – 9 . doi: 10.1038/s41586-018-0579-z OpenUrl CrossRef PubMed Pedersen CB , Bybjerg-Grauholm J , Pedersen MG , et al. The iPSYCH2012 case-cohort sample: new directions for unravelling genetic and environmental architectures of severe mental disorders . Mol Psychiatry . 2018 ; 23 : 6 – 14 . doi: 10.1038/mp.2017.196 OpenUrl CrossRef PubMed ↵ The UK Biobank . Health-related outcomes data . 2022 . https://www.ukbiobank.ac.uk/enable-your-research/about-our-data/health-related-outcomes-data (accessed 26 June 2023) ↵ Smoller JW . The use of electronic health records for psychiatric phenotyping and genomics . Am J Med Genet B Neuropsychiatr Genet . 2018 ; 177 : 601 – 12 . doi: 10.1002/ajmg.b.32548 OpenUrl CrossRef PubMed ↵ HDR UK Phenotype Library . A Reference Catalogue of Human Diseases . http://phenotypes.healthdatagateway.org/ (accessed 4 December 2024) ↵ Karystianis G , Sheppard T , Dixon WG , et al. Modelling and extraction of variability in free-text medication prescriptions from an anonymised primary care electronic medical record research database . BMC Med Inform Decis Mak . 2016 ; 16 : 18 . doi: 10.1186/s12911-016-0255-x OpenUrl CrossRef ↵ Pazzagli L , Brandt L , Linder M , et al. Methods for constructing treatment episodes and impact on exposure-outcome associations . Eur J Clin Pharmacol . 2020 ; 76 : 267 – 75 . doi: 10.1007/s00228-019-02780-4 OpenUrl CrossRef PubMed ↵ Young JC , Conover MM , Funk MJ . Measurement error and misclassification in electronic medical records: methods to mitigate bias . Curr Epidemiol Rep . 2018 ; 5 : 343 – 56 . doi: 10.1007/s40471-018-0164-x OpenUrl CrossRef PubMed ↵ Fabbri C , Hagenaars SP , John C , et al. Genetic and clinical characteristics of treatment-resistant depression using primary care records in two UK cohorts . Mol Psychiatry . 2021 ; 26 : 3363 – 73 . doi: 10.1038/s41380-021-01062-9 OpenUrl CrossRef PubMed ↵ Darke P , Cassidy S , Catt M , et al. Curating a longitudinal research resource using linked primary care EHR data-a UK Biobank case study . J Am Med Inform Assoc . 2022 ; 29 : 546 – 52 . doi: 10.1093/jamia/ocab260 OpenUrl CrossRef PubMed ↵ Lo CWH , Gillett AC , Iveson MH , et al. Antidepressant switching as a proxy phenotype for drug non-response: investigating clinical, demographic and genetic characteristics . Biological Psychiatry Global Open Science . 2025 ; 100502 . doi: 10.1016/j.bpsgos.2025.100502 OpenUrl CrossRef PubMed ↵ Grzenda A , Widge AS . Electronic health records and stratified psychiatry: bridge to precision treatment? Neuropsychopharmacology . 2024 ; 49 : 285 – 90 . doi: 10.1038/s41386-023-01724-y OpenUrl CrossRef PubMed ↵ Fabbri C . Treatment-resistant depression: role of genetic factors in the perspective of clinical stratification and treatment personalisation . Mol Psychiatry . 2025 ; 30 : 2210 – 8 . doi: 10.1038/s41380-025-02899-0 OpenUrl CrossRef PubMed ↵ Xu B , Forthman KL , Kuplicki R , et al. Genetic Correlates of Treatment-Resistant Depression . JAMA Psychiatry . 2025 ; 82 : 505 – 13 . doi: 10.1001/jamapsychiatry.2024.4825 OpenUrl CrossRef PubMed ↵ Lundberg J , Cars T , Lööv S-Å , et al. Association of Treatment-Resistant Depression With Patient Outcomes and Health Care Resource Utilization in a Population-Wide Study . JAMA Psychiatry . 2023 ; 80 : 167 – 75 . doi: 10.1001/jamapsychiatry.2022.3860 OpenUrl CrossRef PubMed ↵ Karageorgiou V , Casanova F , O’Loughlin J , et al. Body mass index and inflammation in depression and treatment-resistant depression: a Mendelian randomisation study . BMC Med . 2023 ; 21 : 355 . doi: 10.1186/s12916-023-03001-7 OpenUrl CrossRef PubMed ↵ Koch E , Pardiñas AF , O’Connell KS , et al. How Real-World Data Can Facilitate the Development of Precision Medicine Treatment in Psychiatry . Biol Psychiatry . 2024 ; 96 : 543 – 51 . doi: 10.1016/j.biopsych.2024.01.001 OpenUrl CrossRef PubMed ↵ Abbasizanjani H , Torabi F , Bedston S , et al. Harmonising electronic health records for reproducible research: challenges, solutions and recommendations from a UK-wide COVID-19 research collaboration . BMC Med Inform Decis Mak . 2023 ; 23 : 8 . doi: 10.1186/s12911-022-02093-0 OpenUrl CrossRef PubMed ↵ Koch E , Smart S , Einarsson G , et al. Recommendations for defining treatment outcomes in major psychiatric disorders using real-world data . Lancet Psychiatry . 2025 ;S2215-0366(25)00061-6. doi: 10.1016/S2215-0366(25)00061-6 OpenUrl CrossRef ↵ R Core Team . R: A Language and Environment for Statistical Computing . 2023 . ↵ English Prescribing Dataset (EPD) - Open Data Portal . https://opendata.nhsbsa.net/dataset/english-prescribing-data-epd (accessed 29 April 2025) ↵ Herrett E , Gallagher AM , Bhaskaran K , et al. Data Resource Profile: Clinical Practice Research Datalink (CPRD) . Int J Epidemiol . 2015 ; 44 : 827 – 36 . doi: 10.1093/ije/dyv098 OpenUrl CrossRef PubMed ↵ Finer S , Martin HC , Khan A , et al. Cohort Profile: East London Genes & Health (ELGH), a community-based population genomics and health study in British Bangladeshi and British Pakistani people . Int J Epidemiol . 2020 ; 49 : 20 – 21i . doi: 10.1093/ije/dyz174 OpenUrl CrossRef ↵ Pazzagli L , Andersen M , Sessa M . Pharmacological and epidemiological considerations while constructing treatment episodes using observational data: A simulation study . Pharmacoepidemiol Drug Saf . 2022 ; 31 : 55 – 60 . doi: 10.1002/pds.5366 OpenUrl CrossRef PubMed ↵ Lo CWH , Fei Y , Cheung BMY . Cardiovascular Outcomes in Trials of New Antidiabetic Drug Classes . Card Fail Rev . 2021 ; 7 : e04 . doi: 10.15420/cfr.2020.19 OpenUrl CrossRef ↵ Wolf A , Dedman D , Campbell J , et al. Data resource profile: Clinical Practice Research Datalink (CPRD) Aurum . International Journal of Epidemiology . 2019 ; 48 : 1740 – 1740g . doi: 10.1093/ije/dyz034 OpenUrl CrossRef PubMed ↵ Clinical Practice Research Datalink . CPRD Aurum December 2023. 2023 . ↵ Wong WLE , Fabbri C , Laplace B , et al. The Effects of CYP2C19 Genotype on Proxies of SSRI Antidepressant Response in the UK Biobank . Pharmaceuticals (Basel ) . 2023 ; 16 : 1277 . doi: 10.3390/ph16091277 OpenUrl CrossRef ↵ Franklin CE , Achtyes E , Altinay M , et al. The genetics of severe depression . Mol Psychiatry . Published Online First: 15 October 2024 . doi: 10.1038/s41380-024-02731-1 OpenUrl CrossRef ↵ Choi L , Beck C , McNeer E , et al. Development of a System for Postmarketing Population Pharmacokinetic and Pharmacodynamic Studies Using Real-World Data From Electronic Health Records . Clin Pharma and Therapeutics . 2020 ; 107 : 934 – 43 . doi: 10.1002/cpt.1787 OpenUrl CrossRef ↵ McInnes G , Altman RB . Drug Response Pharmacogenetics for 200,000 UK Biobank Participants . Pac Symp Biocomput . 2021 ; 26 : 184 – 95 . OpenUrl PubMed ↵ Carr DF , Turner RM , Pirmohamed M . Pharmacogenomics of anticancer drugs: Personalising the choice and dose to manage drug response . Br J Clin Pharmacol . 2021 ; 87 : 237 – 55 . doi: 10.1111/bcp.14407 OpenUrl CrossRef PubMed ↵ Pye SR , Sheppard T , Joseph RM , et al. Assumptions made when preparing drug exposure data for analysis have an impact on results: An unreported step in pharmacoepidemiology studies . Pharmacoepidemiol Drug Saf . 2018 ; 27 : 781 – 8 . doi: 10.1002/pds.4440 OpenUrl CrossRef PubMed ↵ Jagadeesan KK , Grant J , Griffin S , et al. PrAna: an R package to calculate and visualize England NHS primary care prescribing data . BMC Med Inform Decis Mak . 2022 ; 22 : 5 . doi: 10.1186/s12911-021-01727-z OpenUrl CrossRef PubMed ↵ Colón-Ruiz C , Fitzgerald T , Segura-Bedmar I , et al. Automated Extraction and Classification of Drug Prescriptions in Electronic Health Records: Introducing the PRESNER Pipeline . 2023 . ↵ Kormilitzin A , Vaci N , Liu Q , et al. Med7: A transferable clinical natural language processing model for electronic health records . Artif Intell Med . 2021 ; 118 : 102086 . doi: 10.1016/j.artmed.2021.102086 OpenUrl CrossRef PubMed ↵ Soguero-Ruiz C , Hindberg K , Rojo-Alvarez JL , et al. Support Vector Feature Selection for Early Detection of Anastomosis Leakage From Bag-of-Words in Electronic Health Records . IEEE J Biomed Health Inform . 2016 ; 20 : 1404 – 15 . doi: 10.1109/JBHI.2014.2361688 OpenUrl CrossRef PubMed ↵ Jazayeri A , Liang OS , Yang CC . Imputation of Missing Data in Electronic Health Records Based on Patients’ Similarities . J Healthc Inform Res . 2020 ; 4 : 295 – 307 . doi: 10.1007/s41666-020-00073-5 OpenUrl CrossRef PubMed ↵ Wells BJ , Chagin KM , Nowacki AS , et al. Strategies for handling missing data in electronic health record derived data . EGEMS (Wash DC ) . 2013 ; 1 : 1035 . doi: 10.13063/2327-9214.1035 OpenUrl CrossRef PubMed View the discussion thread. Back to top Previous Next Posted October 08, 2025. Download PDF Supplementary Material Data/Code Email Thank you for your interest in spreading the word about medRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following T-Rx: A toolbox for reproducible processing of prescriptions (Rx) from electronic health records Message Subject (Your Name) has forwarded a page to you from medRxiv Message Body (Your Name) thought you would like to see this page from the medRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share T-Rx: A toolbox for reproducible processing of prescriptions (Rx) from electronic health records Chris Wai Hang Lo , Dale Handley , Oliver Pain , Michelle Kamp , Alexandra C. Gillett , Matthew H. Iveson , Chiara Fabbri , Katherine G. Young , AMBER Research Team , Cathryn M. Lewis medRxiv 2025.10.07.25336002; doi: https://doi.org/10.1101/2025.10.07.25336002 Share This Article: Copy Citation Tools T-Rx: A toolbox for reproducible processing of prescriptions (Rx) from electronic health records Chris Wai Hang Lo , Dale Handley , Oliver Pain , Michelle Kamp , Alexandra C. Gillett , Matthew H. Iveson , Chiara Fabbri , Katherine G. Young , AMBER Research Team , Cathryn M. Lewis medRxiv 2025.10.07.25336002; doi: https://doi.org/10.1101/2025.10.07.25336002 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Health Informatics Subject Areas All Articles Addiction Medicine (568) Allergy and Immunology (863) Anesthesia (300) Cardiovascular Medicine (4438) Dentistry and Oral Medicine (444) Dermatology (383) Emergency Medicine (608) Endocrinology (including Diabetes Mellitus and Metabolic Disease) (1509) Epidemiology (15229) Forensic Medicine (30) Gastroenterology (1125) Genetic and Genomic Medicine (6600) Geriatric Medicine (668) Health Economics (997) Health Informatics (4538) Health Policy (1368) Health Systems and Quality Improvement (1613) Hematology (542) HIV/AIDS (1264) Infectious Diseases (except HIV/AIDS) (15919) Intensive Care and Critical Care Medicine (1103) Medical Education (623) Medical Ethics (147) Nephrology (667) Neurology (6600) Nursing (346) Nutrition (998) Obstetrics and Gynecology (1144) Occupational and Environmental Health (957) Oncology (3333) Ophthalmology (974) Orthopedics (369) Otolaryngology (420) Pain Medicine (436) Palliative Medicine (130) Pathology (663) Pediatrics (1693) Pharmacology and Therapeutics (691) Primary Care Research (711) Psychiatry and Clinical Psychology (5447) Public and Global Health (9233) Radiology and Imaging (2199) Rehabilitation Medicine and Physical Therapy (1370) Respiratory Medicine (1196) Rheumatology (593) Sexual and Reproductive Health (712) Sports Medicine (530) Surgery (712) Toxicology (99) Transplantation (289) Urology (265) (function(){function c(){var b=a.contentDocument||a.contentWindow.document;if(b){var d=b.createElement('script');d.innerHTML="window.__CF$cv$params={r:'a0106f202b78aa64',t:'MTc3OTY2OTA1Mw=='};var a=document.createElement('script');a.src='/cdn-cgi/challenge-platform/scripts/jsd/main.js';document.getElementsByTagName('head')[0].appendChild(a);";b.getElementsByTagName('head')[0].appendChild(d)}}if(document.body){var a=document.createElement('iframe');a.height=1;a.width=1;a.style.position='absolute';a.style.top=0;a.style.left=0;a.style.border='none';a.style.visibility='hidden';document.body.appendChild(a);if('loading'!==document.readyState)c();else if(window.addEventListener)document.addEventListener('DOMContentLoaded',c);else{var e=document.onreadystatechange||function(){};document.onreadystatechange=function(b){e(b);'loading'!==document.readyState&&(document.onreadystatechange=e,c())}}}})();
Text is read by the "Ask this paper" AI Q&A widget below.
Extraction quality varies by source — PMC NXML preserves structure
cleanly, OA-HTML may include some navigation residue, and OA-PDF can
have broken hyphenation. The publisher copy
(via DOI)
is the canonical version.