Rapid Whole Genome Characterization of High-Risk Pathogens Using Long-Read Sequencing to Identify Potential Healthcare Transmission

preprint OA: closed
📄 Open PDF Full text JSON View at publisher

Abstract

Objective Routine use of whole genome sequencing (WGS) has been shown to help identify transmission of pathogens causing healthcare-associated infections (HAIs). However, the current gold standard of short-read, Illumina-based WGS is labor and time-intensive. In light of recent improvements in long-read Oxford Nanopore Technologies (ONT) sequencing, we sought to establish a low resource utilization approach capable of providing accurate WGS-based comparisons of HAI pathogens within a time frame allowing for infection prevention and control (IPC) interventions. Methods WGS was prospectively performed on antimicrobial-resistant pathogens at increased risk of potential healthcare transmission using the ONT MinION sequencer with R10.4.1 flow cells and Dorado basecalling algorithm. Potential transmission was assessed via Ridom SeqSphere+ for core genome multilocus sequence typing and MINTyper for reference-based core genome single nucleotide polymorphisms using previously published cut-off values. The accuracy of our ONT pipeline was determined relative to Illumina-based WGS data generated from the same genomic DNA sample. Results Over a six-month period, 242 bacterial isolates from 216 patients were sequenced by a single operator. Compared to the Illumina gold-standard data, our ONT pipeline achieved a Q score of 60 for assembled genomes, even with a coverage rate of as low as 40X. The mean time from initiating DNA extraction to complete genetic analysis was 2 days (IQR 2-3.25 days). We identified five potential transmission clusters comprising 21 isolates (8.7% of all sequenced strains). Combining ONT WGS data with epidemiological data, >70% (15/21) of the isolates originated from patients with potential healthcare transmission links. Conclusions Via a stand-alone ONT pipeline, we detected potentially transmitted HAI pathogens rapidly and accurately, aligning closely with epidemiological data. Our low-resource method has the potential to assist in the efficient detection and deployment of preventative measures against HAI transmission.
Full text 42,214 characters · extracted from preprint-html · click to expand
Rapid Whole Genome Characterization of High-Risk Pathogens Using Long-Read Sequencing to Identify Potential Healthcare Transmission | medRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-P4HH5NV'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search Rapid Whole Genome Characterization of High-Risk Pathogens Using Long-Read Sequencing to Identify Potential Healthcare Transmission Chin-Ting Wu , William C. Shropshire , Micah M Bhatti , Sherry Cantu , Israel K Glover , Selvalakshmi Selvaraj Anand , Xiaojun Liu , Awdhesh Kalia , View ORCID Profile Todd J. Treangen , Roy F Chemaly , Amy Spallone , Samuel Shelburne doi: https://doi.org/10.1101/2024.08.19.24312266 Chin-Ting Wu 1 Graduate Program in Diagnostic Genetics and Genomics, School of Health Professions, MD Anderson Cancer Center, University of Texas , Houston, TX, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site William C. Shropshire 2 Department of Infectious Diseases, Infection Control, and Employee Health, The University of Texas MD Anderson Cancer Center , Houston, TX Find this author on Google Scholar Find this author on PubMed Search for this author on this site Micah M Bhatti 3 Department of Laboratory Medicine, The University of Texas MD Anderson Cancer Center , Houston, TX Find this author on Google Scholar Find this author on PubMed Search for this author on this site Sherry Cantu 4 Infection Control, Chief Quality Office, The University of Texas MD Anderson Cancer Center , Houston, TX Find this author on Google Scholar Find this author on PubMed Search for this author on this site Israel K Glover 5 Department of Genomic Medicine, The University of Texas MD Anderson Cancer Center , Houston, TX Find this author on Google Scholar Find this author on PubMed Search for this author on this site Selvalakshmi Selvaraj Anand 6 PhD Program in Synthetic Biology Institute Systems, Synthetic, and Physical Biology, Rice University , Houston, TX Find this author on Google Scholar Find this author on PubMed Search for this author on this site Xiaojun Liu 1 Graduate Program in Diagnostic Genetics and Genomics, School of Health Professions, MD Anderson Cancer Center, University of Texas , Houston, TX, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site Awdhesh Kalia 1 Graduate Program in Diagnostic Genetics and Genomics, School of Health Professions, MD Anderson Cancer Center, University of Texas , Houston, TX, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site Todd J. Treangen 7 Department of Computer Science, Rice University , Houston, TX 8 Department of Bioengineering, Rice University , Houston, TX Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Todd J. Treangen Roy F Chemaly 2 Department of Infectious Diseases, Infection Control, and Employee Health, The University of Texas MD Anderson Cancer Center , Houston, TX Find this author on Google Scholar Find this author on PubMed Search for this author on this site Amy Spallone 2 Department of Infectious Diseases, Infection Control, and Employee Health, The University of Texas MD Anderson Cancer Center , Houston, TX 4 Infection Control, Chief Quality Office, The University of Texas MD Anderson Cancer Center , Houston, TX Find this author on Google Scholar Find this author on PubMed Search for this author on this site Samuel Shelburne 2 Department of Infectious Diseases, Infection Control, and Employee Health, The University of Texas MD Anderson Cancer Center , Houston, TX Find this author on Google Scholar Find this author on PubMed Search for this author on this site For correspondence: sshelburne{at}mdanderson.org Abstract Full Text Info/History Metrics Supplementary material Data/Code Preview PDF Abstract Objective Routine use of whole genome sequencing (WGS) has been shown to help identify transmission of pathogens causing healthcare-associated infections (HAIs). However, the current gold standard of short-read, Illumina-based WGS is labor and time-intensive. In light of recent improvements in long-read Oxford Nanopore Technologies (ONT) sequencing, we sought to establish a low resource utilization approach capable of providing accurate WGS-based comparisons of HAI pathogens within a time frame allowing for infection prevention and control (IPC) interventions. Methods WGS was prospectively performed on antimicrobial-resistant pathogens at increased risk of potential healthcare transmission using the ONT MinION sequencer with R10.4.1 flow cells and Dorado basecalling algorithm. Potential transmission was assessed via Ridom SeqSphere+ for core genome multilocus sequence typing and MINTyper for reference-based core genome single nucleotide polymorphisms using previously published cut-off values. The accuracy of our ONT pipeline was determined relative to Illumina-based WGS data generated from the same genomic DNA sample. Results Over a six-month period, 242 bacterial isolates from 216 patients were sequenced by a single operator. Compared to the Illumina gold-standard data, our ONT pipeline achieved a Q score of 60 for assembled genomes, even with a coverage rate of as low as 40X. The mean time from initiating DNA extraction to complete genetic analysis was 2 days (IQR 2-3.25 days). We identified five potential transmission clusters comprising 21 isolates (8.7% of all sequenced strains). Combining ONT WGS data with epidemiological data, >70% (15/21) of the isolates originated from patients with potential healthcare transmission links. Conclusions Via a stand-alone ONT pipeline, we detected potentially transmitted HAI pathogens rapidly and accurately, aligning closely with epidemiological data. Our low-resource method has the potential to assist in the efficient detection and deployment of preventative measures against HAI transmission. Introduction Healthcare-associated infections (HAIs) cause tens of thousands of deaths and cost around $3 and $27 billion annually in England and the U.S. respectively [ 1 ]. Infection prevention and control (IPC) teams are critical to mitigating transmission of HAI pathogens, but typical IPC methodologies heavily rely on the intuition of infection control professionals, are time-consuming, and can either over- or under-identify outbreaks [ 2 ]. As whole genome sequencing (WGS) becomes more affordable and feasible, routine WGS has proven effective in detecting clusters of pathogens not meeting typical IPC criteria for potential HAI transmission. For example, Sundermann et al. found that over 10% of isolates were part of genetically related clusters, and only 44% of the isolates identified to be genetically related would be classified as healthcare-associated transmissions using standard National Healthcare Safety Network (NHSN) criteria [ 3 , 4 ]. Similarly, Australian investigators used WGS to discover that over 30% of patients acquired MDR pathogens from the hospital [ 5 ]. Other research has demonstrated the effectiveness of WGS in distinguishing between methicillin-resistant Staphylococcus aureus (MRSA) outbreaks and pseudo-outbreaks [ 6 , 7 ]. These WGS-based efforts highlight the importance of timely identification of HAI transmission in order to facilitate outbreak control by IPC teams. A major barrier to real-time HAI analysis using WGS is the reliance on highly accurate Illumina short-read sequencing, which requires extensive preparation and batch processing, particularly for institutions that do not have large sequencing facilities [ 8 ]. Oxford Nanopore Technologies (ONT) long-read sequencing is an alternative to Illumina short-read sequencing and generally requires minimal sample preparation, can readily be adapted to varying numbers of strains being sequenced, and provides read lengths that facilitate complete genome assemblies including plasmids [ 9 ]. Whereas ONT previously had unacceptably high error rates for assessing bacterial genetic relatedness, advancements like improved V14 chemistries, double-sensor R10 nanopores, and enhanced consensus basecalling models have significantly increased accuracy [ 10 ]. These improvements suggest that ONT sequencing could be a viable standalone real-time WGS HAI analysis option. In this study, we aimed to establish an ONT-only sequencing pipeline capable of rapidly and accurately producing WGS data to classify the genetic relatedness among potentially transmitted antimicrobial-resistant (AMR) pathogens in a tertiary care cancer hospital. Methods Study population The study took place at the University of Texas MD Anderson Cancer Center (MDACC), a 760-bed tertiary care facility in Houston TX, USA. The study timeframe was from August 2023 to March 2024 and was approved by the MDACC quality improvement institutional review board. To optimize the chances of identifying transmitted MDR pathogens, we studied organisms previously identified at high-risk of HAI transmission, namely MRSA, vancomycin resistant Enterococcus faecium (VREfm), and carbapenem-resistant forms of Enterobacterales , Acinetobacter baumannii and Pseudomonas aeruginosa [ 4 , 11 ]. Screening of the electronic health record was performed twice weekly to identify bacteria of interest with final inclusion being limited to those isolated from patients hospitalized for ≥ 48 hours at the time of infection onset or in patients with extensive recent contact (≤ 30 days) with the MDACC healthcare system (i.e. admitted to the hospital or undergoing an outpatient procedure). Genomic DNA extraction, long-read sequencing, and data analysis Genomic DNA (gDNA) was extracted directly from plates obtained from the MDACC clinical microbiology laboratory using the GenElute™ Bacterial Genomic DNA Kit. Long-read libraries were prepared using the Rapid Barcoding Kit 96 V14 and were sequenced on the MinION device using R10.4.1 flow cells following manufacturer instructions. All pod5 reads were basecalled using Dorado v0.5.1 in super high accuracy mode ( [email protected] ) with a minimum quality score filter of 8 to produce FASTQ files. Dorado v0.5.1 was also used to demultiplex and remove adapters from the sequencing results (GitHub: https://github.com/nanoporetech/dorado ). Long-read assemblies were generated using an in-house Flyest package nanopore consensus polishing pipeline (GitHub: https://github.com/wshropshire/flyest ). AMR gene presence was assessed using AMRFinderPlus v3.11.14 [ 12 ]. Accuracy assessment of stand-alone ONT sequencing Illumina sequencing was performed on 55 (22%) samples using the NextSeq500 platform at the MDACC core sequencing facility with a target sequencing depth of ∼100x. Trimmed Illumina short reads were aligned to the consensus sequence assembled from ONT long reads, and variants indicating potential errors in ONT sequencing or assembly were identified using Snippy v4.6.0 with default variant calling parameters ( https://github.com/tseemann/snippy ). Q score was calculated as: Qscore = -10 * log10 (total number of variants identified by Snippy / genome length assembled by Flyer). We utilized Rasusa v0.8.0 (GitHub: https://github.com/mbhall88/rasusa ) to subsample our ONT FASTQ files to coverages of 100x, 80x, 60x, 40x, and 20x for the 35 strains with at least 100x ONT coverage. The subsampled data were used to generate assembled genomes, and the Illumina data were used to identify potential errors as described above. Measures of genetic relatedness Sequence types (ST) and clonal complexes (CC) were determined by PubMLST databases. To identify strains with sufficient genetic similarity to indicate potential transmission, we conducted a two-step screening with cutoff values for different species derived from previously published studies ( Table 1 ) [ 13 – 16 ]. First, SeqSphere+ software (Ridom SeqSphere+ version 9.0.10) was used for core genome multilocus sequence typing (cgMLST) as the primary screening method. The FASTA file from the ONT assembly was utilized, requiring at least 95% of cgMLST target genes for subsequent analysis. For strains that met the cgMLST cut-off, reference-based single nucleotide polymorphism (SNP) calling was performed using MINTyper version 1.1.0 directly on the FASTQ files generated from Dorado demultiplexing [ 17 ]. For strains where cgMLST schema are currently not available in SeqSphere+ (e.g. Enterobacter cloacae ), genetic relatedness was exclusively assessed using MINTyper with cut-offs derived from previously published data [ 18 ]. View this table: View inline View popup Download powerpoint Table 1: Cutoff values for two-step screening Combination of genetic and epidemiologic data For strains that met our potential transmission genetic cut-off values, transmission likelihood was classified based on previously published definitions with probable transmission assigned to patients who stayed on the same ward with at least 24 hours of overlap, possible transmission being for patients who stayed in the same ward within 60 days without overlap, and unlikely transmission when neither of the above criteria were met [ 5 ]. Figure 1 illustrates the entire workflow of our ONT sequencing process. Download figure Open in new tab Figure 1. Workflow of stand-alone real-time ONT sequencing pipeline. Results Standalone ONT sequencing data accuracy From August 2023 to March 2024, we performed ONT sequencing on 242 unique clinical isolates from 216 patients with a species breakdown of 83 MRSA (33%), 41 VREfm (17%), 37 P. aeruginosa (15%), 31 E. coli (13%), 26 K. pneumoniae (11%), and 24 other species (10%) (Table S1). Once we had performed ONT sequencing of at least ten isolates of the five top species, we performed parallel Illumina sequencing on the same genomic DNA. We mapped the Illumina reads to the genomes assembled with ONT data to determine the number of SNPs/ insertion-deletion (INDELs) (i.e. errors) in the ONT assemblies. The median number of SNPs was 1, interquartile range 1 to 3 (IQR): 0-2 and the median number of (INDELs) was 3 (IQR: 1-5) ( Figure 2A ). The average Q score for the ONT-assembled genomes was 60.7 (IQR: 56.9-64.5). We observed that 87% of SNPs were transitions, with 42% being T to C and 33% being A to G ( Figure 2B ). The outlier in Figure 2A , a ST584 K. pneumoniae sample, exhibited 60 SNPs, with 31 being A to G and 29 being T to C. The Integrative Genomics Viewer of an example SNP is shown in Figure 2C . We conclude that relative to the Illumina gold standard, the ONT sequencing was highly accurate although still with a low level of A to G and T to C transitions likely due to poor modeling of unique methylation motifs [ 19 ]. Download figure Open in new tab Figure 2: Standalone ONT sequencing data accuracy validation (A) Number of variants (i.e. errors) in standalone ONT pipeline vs Illumina data (n = 55 strains). Errors are classified into single nucleotide polymorphisms (SNPs, left) and insertion/deletions (INDELs, right). Colors indicate strain species shown in legend. (B) SNPs were classified as transitions (light-blue) or transversions (pink) with exact genetic variation shown on X-axis. Numbers refer to the percentage of total SNPs made up by each genetic variation. (C) Example of SNP error site. Upper part of panel shows Illumina reads mapped to ONT assembly where 100% Illumina reads map to referent (i.e., Adenine, A) as an alternative allele (i.e., Guanine, G) and lower panel indicates ONT alignment with approximately a 50% A/G mapping. (D) Impact of sequencing coverage on error rates with ONT sequencing depth used to generate ONT assembly shown on X-axis and total number of errors shown on Y-axis (median and IQR are shown). *** P -value < 0.001. NS = no statistical difference for indicated Pairwise Wilcoxon Rank-Sum test results. Higher sequencing coverage and coverage depth generally improves assembly accuracy by reducing the likelihood of misassembly or missing sequences [ 20 ]. However, targeting lower coverage depth per sample allows for sequencing more samples per flow cell, which can be cost-effective [ 20 ]. Thus, we next aimed to optimize sequencing coverage to balance cost-effectiveness and accuracy. We used Rasusa to subsample the ONT files to achieve 100X, 80X, 60X, 40X, and 20X coverage. We found that the total number of variants was significant higher at 20X coverage relative to 100X coverage (P < 0.001 by Kruskal-Wallis test followed by pairwise Wilcoxon rank-sum test), but no significant differences in total variants were observed for coverages ≥ 40X ( Figure 2D ). Execution of our standalone ONT sequencing workflow Using our ONT sequencing workflow ( Figure 1 ), over a 26-week study period, we screened 494 positive AMR culture reports. Of these, 246 met our inclusion criteria with three strains excluded due to discrepancies between species ID in the clinical microbiology laboratory and by sequencing analysis whereas one strain was excluded due to sequencing failure leaving a total of 242 isolates for analysis. The most common isolate source was blood (36%, 88/242), followed by tissue/wound/body fluid (24%, 58/242), urinary tract (20%, 48/242), and respiratory tract (18%, 44/242). On average, 10 strains were sequenced per week. The fastest time from initiating gDNA extraction to completing data analysis was one day, with a mean duration of two days (IQR: 2-3.25 days). Detailed timelines of the workflow and sequencing quality control data are provided in Figure S1 and Table S1. Overview of major sequence types amongst sequenced pathogens Phylogenetic trees for each of the five major species are shown in Figures S2-6. Among the 31 E. coli isolates, half belonged to ST131 (n=16, 51.6%) followed by ST167 (n=4, 13%) and ST361 (n=2, 7%). The 26 K. pneumoniae isolates displayed diverse STs with ST258 being the most common (n=4, 15%), followed by ST45 (n=3, 12%) and ST147 (n=3, 12%). For 37 P. aeruginosa strains, over 19% (n=7) were identified as ST633, while other isolates had unique sequence types. ST117 (25 of 41 total isolates, 61%) dominated the VREfm population followed by ST80 (n=6, 15%). Most MRSA isolates (n = 83) were from CC8 (n=38, 46%), CC5 (n=30, 36%), and CC30 (n=8, 10%). A heat map of major genes mediating acquired β-lactam resistance is shown in Figure S7. Except for the large numbers of P. aeruginosa ST633 strains, our AMR pathogen epidemiology generally aligned with predominant STs/CCs reported globally [ 21 – 25 ]. Characterization of potential transmission using genetic relatedness assessment Using our two-step screening process, we identified 21 isolates (8.7%) meeting the criteria for possible transmission, forming five genetically related clusters ( Figure 3 ). Among these, three clusters were from ST117 VREfm and one from ST80 VREfm, with 46% of VREfm strains (19/41) meeting the genetic relatedness cut-off for possible transmission. Another cluster involved two ST45 K. pneumoniae isolates. Table S2 details the detected clusters. To visualize potential healthcare transmissions, we used patient-ward-movement timelines to integrate genomic and epidemiological data. Figure 4 shows the largest ST117 VREfm genetic-related cluster, comprising 12 unique patients. Within this cluster, seven patients (A, C, D, E, H, J, and L) shared overlapping stays in ward 1 and ward 4 for at least one day, indicating a probable transmission. Two patients (B and I) were on ward 2 at different times but within 60 days, suggesting a possible transmission. However, three patients (F, G, and K) did not have overlapping admissions or locations. Potential epidemiological links were present in >70% (15/21) of isolates that met our genetically related cut-off values. Download figure Open in new tab Figure 3: Genetic-related cluster network found in our standalone ONT sequence pipeline Each dot represents one sequenced isolate, color-coded by bacterial species. The network plot was visualized using Gephi. Dots on the outer circle without lines indicate isolates above the cutoff values, while dots in the inner circle, connected by lines, indicate strains below the cutoff values, suggesting possible transmission. Download figure Open in new tab Figure 4. Patient-Ward-Movement-Timelines for VREfm Cluster IV Twelve unique patients in cluster IV are listed on the left. Day of study is shown on the x-axis. Colored boxes in the figure indicate the time-ward-movement timeline for the 12 patients with the individual colors as shown in the legend on the right. Black circle dots represent sample collection date. Use of ONT data to assess potential plasmid transmission Given the power of long-read ONT sequencing to assess plasmid composition [ 26 ], we also sought to determine whether our ONT sequencing pipeline could assess the possibility of plasmid transfer among different bacterial strains/species. We focused on plasmids containing bla NDM genes, which encode New Delhi metallo-β-lactamase enzymes conferring resistance to a broad range of β-lactam antibiotics [ 27 ]. Table S3 lists eight isolates carrying bla NDM genes from seven unique patients, with two isolates originating from the same patient (both from the same urine culture but identified as strains with slightly different AMR profiles). Given that all detected bla NDM-5 genes were carried by IncF type replicon plasmids, we analyzed pairwise SNP distances to study potential horizontal plasmid transfer (Table S4). Only plasmids from the two E. coli strains from the same sample fell below our predefined cutoff value (<15 SNPs/100 Kb) [ 28 ]. Discussion There is increasing recognition that analysis of genetic relatedness amongst bacteria causing infections in healthcare settings could substantially add to IPC efforts to mitigate pathogen acquisition [ 1 , 18 ]. However, widespread use of such approaches is currently limited by a host of factors including difficulty providing timely data in a cost-effective fashion [ 29 ]. Herein, we demonstrate that a low-resource infrastructure using an ONT sequencing pipeline can produce accurate WGS information in a time frame commiserate with impactful IPC interventions. Given its flexibility and low instrumentation requirements, ONT sequencing has long been considered as potentially impactful on infectious diseases surveillance such as its use in field-based sequencing during the 2015 Ebola outbreak [ 30 ]. However, prior to the release of the R10.4.1 flow cells and Dorado basecalling algorithm, the high ONT error rate meant that it was not sufficiently accurate to determine whether bacteria were potentially part of a transmission network [ 9 , 20 ]. When analyzing a diverse array of AMR pathogens, we found that stand-alone ONT sequencing generated complete genomes that generally varied from the gold standard Illumina by only 1-2 SNPs. These data are consistent with studies emerging from other investigations using recent ONT pipelines which generally have analyzed historical cohorts [ 9 , 20 , 31 ]. Moreover, our data were generated prospectively and analyzed by a single operator indicating the low resource utilization of our approach. In light of the tremendous impact of HAI pathogens, modeling has indicated potential cost-savings with routine WGS [ 32 ]. However, the high costs of a core sequencing facility capable of generating Illumina data within a time frame needed for effective IPC efforts may limit its application to research centers [ 4 , 5 ]. We envision that ONT sequencing could be effectively utilized by a broad variety of biomedical facilities to limit HAI transmission, effectively democratizing the integration of WGS into IPC efforts [ 9 ]. Previously, routine sequencing of a diverse array of HAI pathogens showed that particular species were more likely to meet genetic cut-offs for being potentially healthcare transmitted, which caused us to focus on a high-risk group of bacterial pathogens [ 4 ]. Still, our finding that 8.7% of such isolates met the criteria for potential transmission was lower than the 10.8% reported in the Pittsburgh, U.S.A. based study [ 4 ] or the 30% reported in a recent investigation from Australia [ 5 ]. One potential explanation is that with our relatively low number of samples, we lacked the power to fully identify clusters. Additionally, we almost exclusively analyzed clinical infection samples whereas many of the clusters identified in the Australian study came from active colonization surveillance, and thus we may have missed transmission instances that did not result in clinical disease. Finally, given the highly immunocompromised nature of our patients, robust IPC efforts at our hospital, such as routine gloves and masking when entering the rooms of certain types of patients, may mitigate transmission. In spite of some differences, a striking consistent finding between our studies and others is the very high rates of genetic relatedness amongst VREfm which was 46% in our study, 36% in the Pittsburgh study, and 92% in the Australian investigation [ 4 , 5 ]. These findings are not limited to a single ST and suggest that much work remains to be done regarding limiting transmission of VREfm in healthcare settings. Conversely, unlike several recent studies, we found no clusters of closely related MRSA strains despite sequencing 80+ isolates [ 7 , 13 ]. Thus, our data also suggest that WGS resources could be targeted using adaptive strategies dependent upon local findings rather than sequencing all drug-resistant pathogens. In summary, we demonstrate that a low-resource, stand-alone ONT sequencing platform shows promise for real-time monitoring of healthcare-associated transmission and outbreak detection. Integration of such an approach into IPC efforts could assist with limiting HAIs in a wide variety of healthcare settings. Data Availability All data produced in the present study are available upon reasonable request to the authors Funding information Core grant CA016672 (Advanced Technology Genomics Core - ATGC) and NIH grant 1S10OD024977-01 provided funding for the ATGC sequencing facility at MDACC. C.-T.W. was supported by a Peter and Cynthia Hu scholarship. W.C.S. was supported through the National Institute of Allergy and Infectious Diseases (NIAID) T32 AI141349 Training Program in Antimicrobial Resistance. Support for this study was also provided by NIAID grants R21AI151536 and P01AI152999 for S.A.S and T.J.T, and provided by the National Science Foundation (NSF: EF-2126387, IIS-2239114) to T.J.T. Conflicts of interest The authors declare that there are no conflicts of interest. Ethical statement This study received approval from the University of Texas MDACC Quality Improvement Assessment Board (protocol ID no. QIAB-1051). Acknowledgements We thank the members of the MDACC clinical microbiology laboratory for providing isolates in timely fashion. The authors acknowledge the support of the high-performance computing research facility at the University of Texas MDACC for providing computational resources that have contributed to the research results reported in this paper. We acknowledgment some pictures were created with BioRender.com. The authors thank Andrew Yang’s help for scientific writer and editorial assistance. References 1. ↵ Guest , J.F. , et al. , Modelling the annual NHS costs and outcomes attributable to healthcare-associated infections in England . BMJ Open , 2020 . 10 ( 1 ): p. e033367 . OpenUrl Abstract / FREE Full Text 2. ↵ Peacock , S.J. , J. Parkhill , and N.M. Brown , Changing the paradigm for hospital outbreak detection by leading with genomic surveillance of nosocomial pathogens . Microbiology (Reading ), 2018 . 164 ( 10 ): p. 1213 – 1219 . OpenUrl CrossRef 3. ↵ Sundermann , A.J. , et al. , Sensitivity of National Healthcare Safety Network definitions to capture healthcare-associated transmission identified by whole-genome sequencing surveillance . Infect Control Hosp Epidemiol , 2023 . 44 ( 10 ): p. 1663 – 1665 . OpenUrl 4. ↵ Sundermann , A.J. , et al. , Whole-Genome Sequencing Surveillance and Machine Learning of the Electronic Health Record for Enhanced Healthcare Outbreak Detection . Clin Infect Dis , 2022 . 75 ( 3 ): p. 476 – 482 . OpenUrl 5. ↵ Sherry , N.L. , et al. , Multi-site implementation of whole genome sequencing for hospital infection control: A prospective genomic epidemiological analysis . Lancet Reg Health West Pac , 2022 . 23 : p. 100446 . OpenUrl 6. ↵ Talbot , B.M. , et al. , Unsuspected Clonal Spread of Methicillin-Resistant Staphylococcus aureus Causing Bloodstream Infections in Hospitalized Adults Detected Using Whole Genome Sequencing . Clin Infect Dis , 2022 . 75 ( 12 ): p. 2104 – 2112 . OpenUrl CrossRef PubMed 7. ↵ Blane , B. , et al. , Evaluating the impact of genomic epidemiology of methicillin-resistant Staphylococcus aureus (MRSA) on hospital infection prevention and control decisions . Microb Genom , 2024 . 10 ( 4 ). 8. ↵ Maljkovic Berry , I. , et al. , Next Generation Sequencing and Bioinformatics Methodologies for Infectious Disease Research and Public Health: Approaches, Applications, and Considerations for Development of Laboratory Capacity . J Infect Dis , 2020 . 221 ( Suppl 3 ): p. S292 – S307 . OpenUrl 9. ↵ Wagner , G.E. , et al. , Real-Time Nanopore Q20+ Sequencing Enables Extremely Fast and Accurate Core Genome MLST Typing and Democratizes Access to High-Resolution Bacterial Pathogen Surveillance . J Clin Microbiol , 2023 . 61 ( 4 ): p. e0163122 . OpenUrl CrossRef 10. ↵ Zhang , T. , et al. , The newest Oxford Nanopore R10.4.1 full-length 16S rRNA sequencing enables the accurate resolution of species-level microbial community profiling . Appl Environ Microbiol , 2023 . 89 ( 10 ): p. e0060523 . OpenUrl CrossRef 11. ↵ Gorrie , C.L. , et al. , Key parameters for genomics-based real-time detection and tracking of multidrug-resistant bacteria: a systematic analysis . Lancet Microbe , 2021 . 2 ( 11 ): p. e575 – e583 . OpenUrl 12. ↵ Feldgarden , M. , et al. , AMRFinderPlus and the Reference Gene Catalog facilitate examination of the genomic links among antimicrobial resistance, stress response, and virulence . Sci Rep , 2021 . 11 ( 1 ): p. 12728 . OpenUrl 13. ↵ Kinnevey , P.M. , et al. , Meticillin-resistant Staphylococcus aureus transmission among healthcare workers, patients and the environment in a large acute hospital under non-outbreak conditions investigated using whole-genome sequencing . J Hosp Infect , 2021 . 118 : p. 99 – 107 . OpenUrl 14. Carlsen , L. , et al. , High burden and diversity of carbapenemase-producing Enterobacterales observed in wastewater of a tertiary care hospital in Germany . Int J Hyg Environ Health , 2022 . 242 : p. 113968 . OpenUrl 15. Maechler , F. , et al. , Split k-mer analysis compared to cgMLST and SNP-based core genome analysis for detecting transmission of vancomycin-resistant enterococci: results from routine outbreak analyses across different hospitals and hospitals networks in Berlin , Germany. Microb Genom , 2023 . 9 ( 1 ). 16. ↵ Martak , D. , et al. , Comparison of pulsed-field gel electrophoresis and whole-genome-sequencing-based typing confirms the accuracy of pulsed-field gel electrophoresis for the investigation of local Pseudomonas aeruginosa outbreaks . J Hosp Infect , 2020 . 105 ( 4 ): p. 643 – 647 . OpenUrl 17. ↵ Hallgren , M.B. , et al. , MINTyper: an outbreak-detection method for accurate and rapid SNP typing of clonal clusters with noisy long reads . Biol Methods Protoc , 2021 . 6 ( 1 ): p. bpab008. 18. ↵ Sundermann , A.J. , et al. , Whole-genome sequencing surveillance and machine learning for healthcare outbreak detection and investigation: A systematic review and summary . Antimicrob Steward Healthc Epidemiol , 2022 . 2 ( 1 ): p. e91 . OpenUrl CrossRef 19. ↵ Sanderson , N.D. , et al. , Evaluation of the accuracy of bacterial genome reconstruction with Oxford Nanopore R10.4.1 long-read-only sequencing . Microb Genom , 2024 . 10 ( 5 ). 20. ↵ Sereika , M. , et al. , Oxford Nanopore R10.4 long-read sequencing enables the generation of near-finished bacterial genomes from pure cultures and metagenomes without short-read or reference polishing . Nat Methods , 2022 . 19 ( 7 ): p. 823 – 826 . OpenUrl CrossRef 21. ↵ Turner , N.A. , et al. , Methicillin-resistant Staphylococcus aureus: an overview of basic and clinical research . Nat Rev Microbiol , 2019 . 17 ( 4 ): p. 203 – 218 . OpenUrl CrossRef PubMed 22. Weber , A. , et al. , Increase of vancomycin-resistant Enterococcus faecium strain type ST117 CT71 at Charite - Universitatsmedizin Berlin, 2008 to 2018 . Antimicrob Resist Infect Control , 2020 . 9 ( 1 ): p. 109 . OpenUrl 23. Gual-de-Torrella , A. , et al. , Molecular characterization of a suspected IMP-type carbapenemase-producing Pseudomonas aeruginosa outbreak reveals two simultaneous outbreaks in a tertiary-care hospital . Infect Control Hosp Epidemiol , 2023 . 44 ( 11 ): p. 1801 – 1808 . OpenUrl 24. Luterbach , C.L. , et al. , Transmission of Carbapenem-Resistant Klebsiella pneumoniae in US Hospitals . Clin Infect Dis , 2023 . 76 ( 2 ): p. 229 – 237 . OpenUrl 25. ↵ Mills , E.G. , et al. , A one-year genomic investigation of Escherichia coli epidemiology and nosocomial spread at a large US healthcare network . Genome Med , 2022 . 14 ( 1 ): p. 147 . OpenUrl 26. ↵ Wick , R.R. , et al. , Recovery of small plasmid sequences via Oxford Nanopore sequencing . Microb Genom , 2021 . 7 ( 8 ). 27. ↵ Sakamoto , N. , et al. , Role of Chromosome- and/or Plasmid-Located bla(NDM) on the Carbapenem Resistance and the Gene Stability in Escherichia coli . Microbiol Spectr , 2022 . 10 ( 4 ): p. e0058722 . OpenUrl 28. ↵ Raabe , N.J. , et al. , Real-time genomic epidemiologic investigation of a multispecies plasmid-associated hospital outbreak of NDM-5-producing Enterobacterales infections . Int J Infect Dis , 2024 . 142 : p. 106971 . OpenUrl 29. ↵ Werner , G. , et al. , Taking hospital pathogen surveillance to the next level . Microb Genom , 2023 . 9 ( 4 ). 30. ↵ Quick , J. , et al. , Real-time, portable genome sequencing for Ebola surveillance . Nature , 2016 . 530 ( 7589 ): p. 228 – 232 . OpenUrl CrossRef PubMed 31. ↵ Bogaerts , B. , et al. , Closing the gap: Oxford Nanopore Technologies R10 sequencing allows comparable results to Illumina sequencing for SNP-based outbreak investigation of bacterial pathogens . J Clin Microbiol , 2024 . 62 ( 5 ): p. e0157623 . OpenUrl CrossRef PubMed 32. ↵ Fox , J.M. , N.J. Saunders , and S.H. Jerwood , Economic and health impact modelling of a whole genome sequencing-led intervention strategy for bacterial healthcare-associated infections for England and for the USA . Microb Genom , 2023 . 9 ( 8 ). View the discussion thread. Back to top Previous Next Posted August 20, 2024. Download PDF Supplementary Material Data/Code Email Thank you for your interest in spreading the word about medRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following Rapid Whole Genome Characterization of High-Risk Pathogens Using Long-Read Sequencing to Identify Potential Healthcare Transmission Message Subject (Your Name) has forwarded a page to you from medRxiv Message Body (Your Name) thought you would like to see this page from the medRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share Rapid Whole Genome Characterization of High-Risk Pathogens Using Long-Read Sequencing to Identify Potential Healthcare Transmission Chin-Ting Wu , William C. Shropshire , Micah M Bhatti , Sherry Cantu , Israel K Glover , Selvalakshmi Selvaraj Anand , Xiaojun Liu , Awdhesh Kalia , Todd J. Treangen , Roy F Chemaly , Amy Spallone , Samuel Shelburne medRxiv 2024.08.19.24312266; doi: https://doi.org/10.1101/2024.08.19.24312266 Share This Article: Copy Citation Tools Rapid Whole Genome Characterization of High-Risk Pathogens Using Long-Read Sequencing to Identify Potential Healthcare Transmission Chin-Ting Wu , William C. Shropshire , Micah M Bhatti , Sherry Cantu , Israel K Glover , Selvalakshmi Selvaraj Anand , Xiaojun Liu , Awdhesh Kalia , Todd J. Treangen , Roy F Chemaly , Amy Spallone , Samuel Shelburne medRxiv 2024.08.19.24312266; doi: https://doi.org/10.1101/2024.08.19.24312266 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Health Systems and Quality Improvement Subject Areas All Articles Addiction Medicine (573) Allergy and Immunology (865) Anesthesia (303) Cardiovascular Medicine (4457) Dentistry and Oral Medicine (445) Dermatology (383) Emergency Medicine (610) Endocrinology (including Diabetes Mellitus and Metabolic Disease) (1517) Epidemiology (15244) Forensic Medicine (30) Gastroenterology (1132) Genetic and Genomic Medicine (6620) Geriatric Medicine (669) Health Economics (1002) Health Informatics (4557) Health Policy (1372) Health Systems and Quality Improvement (1615) Hematology (543) HIV/AIDS (1272) Infectious Diseases (except HIV/AIDS) (15936) Intensive Care and Critical Care Medicine (1106) Medical Education (624) Medical Ethics (147) Nephrology (670) Neurology (6634) Nursing (346) Nutrition (999) Obstetrics and Gynecology (1148) Occupational and Environmental Health (957) Oncology (3348) Ophthalmology (980) Orthopedics (369) Otolaryngology (421) Pain Medicine (436) Palliative Medicine (130) Pathology (665) Pediatrics (1696) Pharmacology and Therapeutics (693) Primary Care Research (714) Psychiatry and Clinical Psychology (5463) Public and Global Health (9257) Radiology and Imaging (2210) Rehabilitation Medicine and Physical Therapy (1371) Respiratory Medicine (1198) Rheumatology (598) Sexual and Reproductive Health (716) Sports Medicine (532) Surgery (714) Toxicology (99) Transplantation (289) Urology (265) (function(){function c(){var b=a.contentDocument||a.contentWindow.document;if(b){var d=b.createElement('script');d.innerHTML="window.__CF$cv$params={r:'a0316c67ded2df88',t:'MTc4MDAxNDk3Mg=='};var a=document.createElement('script');a.src='/cdn-cgi/challenge-platform/scripts/jsd/main.js';document.getElementsByTagName('head')[0].appendChild(a);";b.getElementsByTagName('head')[0].appendChild(d)}}if(document.body){var a=document.createElement('iframe');a.height=1;a.width=1;a.style.position='absolute';a.style.top=0;a.style.left=0;a.style.border='none';a.style.visibility='hidden';document.body.appendChild(a);if('loading'!==document.readyState)c();else if(window.addEventListener)document.addEventListener('DOMContentLoaded',c);else{var e=document.onreadystatechange||function(){};document.onreadystatechange=function(b){e(b);'loading'!==document.readyState&&(document.onreadystatechange=e,c())}}}})();

Text is read by the "Ask this paper" AI Q&A widget below. Extraction quality varies by source — PMC NXML preserves structure cleanly, OA-HTML may include some navigation residue, and OA-PDF can have broken hyphenation. The publisher copy (via DOI) is the canonical version.

My notes (saved in your browser only)

Ask this paper AI returns verbatim quotes from the full text · source: preprint-html

Answers must be backed by verbatim quotes from this paper's full text. Hallucinated quotes are dropped automatically; if no verbatim passage answers the question, we say so. How this works

Citation neighborhood (no data yet)

We don't have any in-corpus citations linked to this paper yet. This is a recent paper (2024) — citers typically take a year or two to land, and the OpenAlex reference graph may still be filling in.

Source provenance

europepmc
last seen: 2026-05-20T01:45:00.602351+00:00
unpaywall
last seen: 2026-07-19T06:49:21.617583+00:00