Full text
65,521 characters
· extracted from
preprint-html
· click to expand
Defining the Mycobacterium tuberculosis Pangenome and Suggestions for a New Composite Reference Sequence | bioRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-M677548'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search New Results Defining the Mycobacterium tuberculosis Pangenome and Suggestions for a New Composite Reference Sequence Poonam Chitale , Elissa Ocke , Aubrey R. Odom , Howard Fan , Alexander Henoch , Karla Vasco , Emily C. Fogarty , Courtney Grady , Alexander D. Lemenze , Pradeep Kumar , Shannon Manning , A. Murat Eren , W. Evan Johnson , David Alland doi: https://doi.org/10.1101/2025.11.03.686283 Poonam Chitale 1 Ray V. Lourenco Center for the Study of Emerging and Re-emerging Pathogens, Rutgers University – New Jersey Medical School , Newark, NJ, USA 2 Public Health Research Institute, Rutgers University – New Jersey Medical School , Newark, NJ, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site Elissa Ocke 1 Ray V. Lourenco Center for the Study of Emerging and Re-emerging Pathogens, Rutgers University – New Jersey Medical School , Newark, NJ, USA 2 Public Health Research Institute, Rutgers University – New Jersey Medical School , Newark, NJ, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site Aubrey R. Odom 3 Division of Computational Biomedicine, Boston University School of Medicine and Bioinformatics Program, Boston University , Boston, MA, USA 4 Bioinformatics Program, Boston University , Boston, MA, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site Howard Fan 2 Public Health Research Institute, Rutgers University – New Jersey Medical School , Newark, NJ, USA 5 Center for Data Science, Rutgers University – New Jersey Medical School , Newark, NJ, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site Alexander Henoch 6 Helmholtz Institute for Functional Marine Biodiversity , 26129 Oldenburg, Germany 7 Alfred Wegener Institute, Helmholtz Centre for Polar and Marine Research , 27570 Bremerhaven, German Find this author on Google Scholar Find this author on PubMed Search for this author on this site Karla Vasco 8 Alexander Henoch, A. Murat ErenDepartment of Microbiology , Genetics, and Immunology, Michigan State University , East Lansing, Michigan, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site Emily C. Fogarty 9 Department of Medicine, University of Chicago , Chicago, IL, USA 10 Committee on Microbiology, University of Chicago , Chicago, IL, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site Courtney Grady 1 Ray V. Lourenco Center for the Study of Emerging and Re-emerging Pathogens, Rutgers University – New Jersey Medical School , Newark, NJ, USA 2 Public Health Research Institute, Rutgers University – New Jersey Medical School , Newark, NJ, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site Alexander D. Lemenze 11 Department of Pathology, Immunology and Laboratory Medicine, New Jersey Medical School, Rutgers—The State University of New Jersey , Newark, NJ, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site Pradeep Kumar 1 Ray V. Lourenco Center for the Study of Emerging and Re-emerging Pathogens, Rutgers University – New Jersey Medical School , Newark, NJ, USA 2 Public Health Research Institute, Rutgers University – New Jersey Medical School , Newark, NJ, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site Shannon Manning 8 Alexander Henoch, A. Murat ErenDepartment of Microbiology , Genetics, and Immunology, Michigan State University , East Lansing, Michigan, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site A. Murat Eren 6 Helmholtz Institute for Functional Marine Biodiversity , 26129 Oldenburg, Germany 7 Alfred Wegener Institute, Helmholtz Centre for Polar and Marine Research , 27570 Bremerhaven, German 12 Institute for Chemistry and Biology of the Marine Environment, University of Oldenburg , 26129 Oldenburg, Germany 13 Max Planck Institute for Marine Microbiology , 28359 Bremen, Germany Find this author on Google Scholar Find this author on PubMed Search for this author on this site W. Evan Johnson 1 Ray V. Lourenco Center for the Study of Emerging and Re-emerging Pathogens, Rutgers University – New Jersey Medical School , Newark, NJ, USA 5 Center for Data Science, Rutgers University – New Jersey Medical School , Newark, NJ, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site David Alland 1 Ray V. Lourenco Center for the Study of Emerging and Re-emerging Pathogens, Rutgers University – New Jersey Medical School , Newark, NJ, USA 2 Public Health Research Institute, Rutgers University – New Jersey Medical School , Newark, NJ, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site For correspondence: david.alland{at}rutgers.edu Abstract Full Text Info/History Metrics Supplementary material Preview PDF ABSTRACT Mycobacterium tuberculosis (Mtb) causes tuberculosis (TB), a global disease with diverse clinical and microbiological manifestations. Studies into the biological causes of this phenotypic diversity have been largely limited to a few reference strains. A pangenome approach is likely to provide new insights. Pangenomic tuberculosis studies have been limited the availability of only fragmented genome sequences and error-prone reference genomes. We used a de novo assembly pipeline that generates extremely complete and accurate whole genome sequences to generate 50 closed Mtb genomes across all seven major lineages. We identified 3,377 core gene clusters and 379 accessory clusters. Analysis showed multi-copy core clusters were largely due to gene fragmentation (76%), paralogs (12%), nearly identical gene duplications (4%), or combinations (8%). Sixteen hypervariable regions (HVRs) were identified, including novel paralogs and variable PE/PPE genes. We consolidated these findings into a Pangenome Gene Reference Resource (PGRR) for precision alignment. Our study demonstrates the closed nature of the Mtb pangenome, with most variation in accessory genes and HVRs. The PGRR provides a foundation for improved drug/vaccine target discovery and highlights the need to move beyond the commonly used H37Rv strain to study Mtb genetic and phenotypic diversity. IMPORTANCE Tuberculosis (TB), caused by Mycobacterium tuberculosis , affects millions globally. Genetic differences among Mtb strains have been difficult to resolve due to incomplete genome references. We sequenced and analyzed complete genomes of 50 Mtb strains from all lineages, identifying 16 hypervariable regions and 3,498 core gene clusters whose diversity mostly stemmed from gene fragmentation, paralog duplication and deletion events and differences in the PE/PPE gene family representation. These differences may explain many of the varied clinical manifestations of TB. We created Pangenome Gene Reference Resource to unify genetic data for precise comparison studies to aid in developing new drugs vaccines and other interventions against this disease. INTRODUCTION Tuberculosis (TB) infects an estimated 10.6 million people and causes 1.6 million deaths annually [ 1 ]. TB is caused predominantly by Mycobacterium tuberculosis sensu stricto (Mtb ) , and Mycobacterium africanum [ 2 , 3 ]. Starting in the 1950s, the introduction of TB chemotherapy drove Mtb strains to acquire an expanding repertoire of drug resistance mutations, with several predominant mutations directly related to drug-target interactions and others to changes in metabolic or other cellular functions with more subtle but still measurable effects on drug resistance [ 4 – 12 ]. Slower progress has been made determining the genetic causes of more complex phenotypes such as transmissibility, virulence, or host range. Genetic studies of these phenotypes are likely to benefit from whole genome analysis of large Mtb populations that includes much of the species genetic diversity. The analyses performed so far have relied on error-prone draft genomes and a limited pool of reference strains—most commonly H37Rv, which itself contains annotated errors [ 13 – 15 ] and is further confounded by the highly repetitive PE/PPE family (nearly 10% of the Mtb genome) that resists accurate assembly and annotation [ 16 , 17 ]. Recent advances in long-read sequencing and high-fidelity assembly permit generation of complete, error-free Mtb genomes [ 13 ] which can be leveraged by pangenomic analysis to reveal the entire gene repertoire of the species. The information gained regarding differences in gene representation can potentially explain much of the phenotypic diversity observed in TB including determinants of drug resistance, host- pathogen interactions, virulence and transmissibility. Prior pangenome analyses have been limited by reliance on short-read draft genomes, reference bias, and incomplete representation of all major global lineages [ 18 – 21 ]. Furthermore, critical regions— especially PE/PPE gene loci—are rarely resolved. High-quality, lineage-diverse, closed genomes are therefore essential to map the true diversity of the Mtb pangenome. Here, we present a comprehensive pangenome based on 50 high-fidelity, completely closed Mtb genomes, representing all seven major lineages. We include an M. canettii strain to root evolutionary analyses but exclude it from the pangenome proper. We also introduce the Pangenome Gene Reference Resource (PGRR)—a consolidated reference for all genes and variants found—enabling researchers to precisely align, compare, and annotate new TB genomes. METHODS Details on bacterial strains used, pyazinamidase activity determination, culture, DNA extraction, Nanopore and Illumina whole genome sequencing along with assembly and annotation, lineage SNP and indel identification, phylogenetic and pangenome analysis, and the Pangenome Gene Reference Resource are provided in “Supplementary Methods”. RESULTS Strain collection and sequencing analysis We analyzed 50 highly accurate Mtb genomes generated with Bact-Builder, a bioinformatics pipeline which uses a consensus- based approach to generate highly accurate and complete closed bacterial genome sequences [ 13 ] that span all seven established lineages to describe the Mtb pangenome ( Table 1 ). For comparison and annotation purposes, our analysis included well characterized reference or laboratory strains H37Rv (TMC102 and NR-123), CDC1551, Erdman and HN878. The remaining strains were obtained from either the Tropical Disease Research (TDR)-TB strain bank established by the World Health Organization Special Program for Research and Training in Tropical Disease [ 22 ] or a collection of 20 ‘reference’ strains representing all seven lineages described previously (Table S1) [ 22 , 23 ]. The average genome size was 4,413,145 + 37,039 (mean + SD) base pairs (bp) with an average GC content of 65.61% ( Table 1 ). Annotated genomes had 4,212 to 4,589 coding sequences (CDS) with an average of 4,266.62 + 50.8, of which 829 to 916 were hypothetical CDS’s with an average of 855.24 + 14.1 ( Table 1 ). View this table: View inline View popup Table 1 Pangenome strain lineages and genome size Phylogenetic analysis A wgMLST maximum likelihood tree, that uses specific alleles identified through whole-genome sequencing to create a high-resolution genetic tree, was constructed based on 90,7201 amino acids in 2,891 core genes and 755 accessory genes to define the evolutionary relationships between the 50 strains ( Fig. 1 ). All strains clustered based on their predefined lineage assignments, which were determined using Mykrobe, a bioinformatic tool that uses whole genome sequencing to identify genomic lineages and predict drug resistance [ 24 ]. Of the 50 strains, 23 (46%) belonged to Lineage 4, 9 (18%) to Lineage 2, 5 (10%) to Lineage 3, 6 (12%) to Lineage 1, 1 (2%) to Lineage 7, and 3 (6% each) to Lineages 5 and 6, respectively ( Table 1 ). This distribution broadly mirrors global distribution of lineages where L4 accounts the majority of categorized strains, followed by L1, L2/3, L5/6, and finally L7 [ 25 ] ( Fig. 1 ). Lineages segregated as expected with “modern” (L2, L3 and L4) lineages sharing their most recent ancestors and the “ancient” (L1, L5, 6, and L7) lineages shown as outgroups. However, inclusion of both the core and accessory gene content allowed for a more complete definition of strain relationships. The resulting tree is notable for not producing any low confidence regions unlike other similar trees [ 26 ] confirming the high quality of our sequencing platform and the Bact-Builder assemblies. Download figure Open in new tab Fig 1 Single nucleotide polymorphism-based phylogenetic tree of all M. tuberculosis pangenome strains. A maximum likelihood phylogenetic tree was inferred using the Q.mammal+F+I subsitiution model. The x-axis represents the branch length which corresponds to the evolutionary distance as expected substitutions per site. The tree includes 50 sequences with 907,201 amino-acid sites including 2,585 parsimony informative sites. Analysis revealed expected major M. tuberculosis lineages and relationships among lineages. Overall distribution of strains roughly matches global lineage distribution. Evaluating SNPs and INDELs A comparison of SNPs and INDELs relative to the reference H37Rv strain showed that Lineage 4 strains possessed the fewest number of SNPs and INDELs relative to H37Rv ( Fig. 2A and 2B ; Table 2 ). Lineage 1 had the highest number of SNPs and Lineage 5 had the highest number of INDELs relative to H37Rv ( Fig. 2A and 2B ). This finding was consistent with other studies [ 8 , 27 ] and not surprising since the comparator H37Rv strain was a lineage 4 strain and. Interestingly, a comparison of intra-lineage point mutations revealed that L4 (average SNPs = 1,100) along with L1 (average SNPs = 1,182) had relatively high number of SNPs within their respective lineages compared to the other lineages ( Fig. 2C and 2D ; Tables S2 and S3). Moreover, L4 also harbored the highest relative number of INDELs (mean INDEL count = 966), closely followed by L1 (mean INDEL count = 811) While we observed this greater frequency of intra-lineage point mutations in L1 and L4, these differences were not found to be significant in a pairwise comparison in terms of SNPs ( p = 0.37) or INDELs ( p = 0.06). There were fewer INDELs on average across the lineages compared to SNPs (Table S2 and S3). The largest SNP difference was 3,300 SNPs between H37Rv (L4) and BCCM_083 (L1), and the highest INDEL difference 2,729 INDELs was between TDR 89 (L4) and BCCM_095 (L5) ( Fig. 2A and 2B ). Download figure Open in new tab Fig 2 Summary of SNP and INDEL differences across all strains. a-b. Average SNP (a) and INDEL (b) values for each lineage compared to H37Rv. c-d. Average SNP (c) and INDEL (d) values for each lineage. Lineage 7 has no bar because there was only 1 L7 sample. Means + SD are depicted. Significance determined using pairwise t-tests adjusted using FDR multiple testing correction (* = P < 0.05, ** = P < 0.01, *** = P < 0.001, and **** = P < 0.0001). Nonsignificant values excluded from figure. View this table: View inline View popup Table 2 SNP and INDEL counts of strains relative to H37Rv Pangenome analysis In contrast to a single genome, a pangenome takes into consideration the entire genetic repertoire of a given set of genomes that belong to a single species or a microbial clade [ 28 ]. Gene representations within a pangenome are often referred to as a set of “core” genes” that are present in all genomes within a population and “accessory” genes that occur only in a subset of the population [ 28 – 30 ]. However, the sequence diversity that typically develops among orthologs across a pangenome can blur the boundaries between orthologs that would commonly be thought of as identical based on synteny and high homology levels versus paralogs that may be functionally similar but not identical. Thus, a typical pangenomic analysis identifies de novo “gene families” (rather than individual genes) which are shared between strains based on the sequence homology of each gene against every other gene across genomes. Highly similar members of gene families can be further distributed into “gene clusters” (which will group together genes that are highly similar to one another at the amino acid sequence level) enabling the identification of "core" (e.g., groups of near- identical genes that are present in all strains) and "accessory" gene clusters (e.g., groups of near-identical genes that are absent in at least one strain). This approach completely disregards gene synteny, so similar sequences will be grouped together in the same gene cluster even if they occur in entirely different genomic contexts. While gene synteny information can help distinguish such cases and separate genes that are identical across genomes from those that are paralogous or duplicated genes (i.e., similar sequences that occur in different genomic contexts), this approach may over-split genes that have the same sequence and serve the same function if the surrounding genomic regions have undergone rearrangements, duplications, or deletions. We performed a pangenome analysis of the 50 Mtb strains with Anvi’o, a software platform that performs comparative analyses of multiple genomes and estimates relationships based on gene content, which revealed 3,756 total gene clusters ( Fig. 3A , see Data Availability). Of these, 3,377 were conserved across all strains and classified as core gene clusters. The majority of the core gene clusters (2,820) contained a single gene from each genome (i.e., single-copy core gene clusters), while 557 of core gene clusters included at least two genes from the same genome. Copy number variation across strains resulted in different frequencies of both single and multiple core gene clusters among genomes ( Table 1 ). The remaining 379 gene clusters were classified as accessory: 291 occurred in multiple strains while 88 were strain-specific singletons. Core gene clusters represented 94.1% to 95.5% of genes in any given Mtb genome, averaging 94.7% representation across all genomes. Rarefaction curves and Heap’s law analysis ( Fig. 3B ) indicated that this 50-strain collection captured the majority of Mtb genetic diversity. The extremely small Heap’s law α value (0.0126) confirmed that Mtb has a highly closed pangenome with extensive genetic conservation across clades. Download figure Open in new tab Fig 3 Comparative genomics of of M. tuberculosis strains. (a) The pangenomic analysis of 50 M. tuberculosis strains. Genomes are organized by the SNP tree shown in the top-right of the display and each major lineage colored. (b-d) Breakdown of functional categories of core (b) and accessory genes (c), and the comparison of the enrichment of functions in different gene pools (d). (e) The panel displays the rarefaction curves for the pangenome that summarize 100iterations along with the Heap’s Law fit. We used the NCBI’s Clusters of Orthologous Genes (COGs) database to gain functional insights into the core and accessory gene pool. 824 of 3,377 (24.4%) core gene clusters and 253 of 379 (66.8%) accessory gene clusters had no functional annotation, suggesting that genes of unknown function comprised of a much larger fraction of the accessory gene pool. The new or previously unannotated genes in the accessory gene pool may constitute promising targets for further investigations (see Data Availability). Among those gene clusters with predicted functions, we observed 21 and 18 functional categories for core and accessory gene clusters, respectively. Most gene clusters with functional annotation in the core Mtb pangenome were involved in lipid transport (7.9% of all core gene clusters), coenzyme transport (6.5%), and translation (5.8%) ( Fig. 3C , see Data Availability). In comparison, most gene clusters with functional annotation in the accessory gene pool were involved in coenzyme transport (3.7% of all accessory gene clusters), mobilome (3.2%), and cell wall/membrane biogenesis (2.6%) ( Fig. 3D , see Data Availability). Side-by-side comparison of the enrichment of functional categories in core and accessory gene clusters show that mobilome genes drive the highest difference between the two pools ( Fig. 3E ). The higher number of mobilome genes among accessory gene clusters is most likely due to the known transposon elements that vary greatly between strains of Mtb [ 31 , 32 ]. We sought to characterize the genetic makeup of multicopy gene clusters. We examined eight Mtb strains [BCCM083, BCCM088, BCCM089, BCCM093, BCCM097, BCCM099, BCCM101, and H37Rv] randomly chosen from our 50-strain set and selected 25 examples of core clusters generated by Anvi’o (Table S4). Clusters from this subset could be principally attributed to four categories of events (Table S5): 1) gene fragmentation, 2) multiple paralogs, 3) nearly identical gene duplication and 4) gene clusters caused by a combination of these features. Gene fragmentation occurred when a single open reading frame was split into two or more open reading frames by a premature stop codon followed by a secondary start codon situated downstream of the initial stop codon causing a gene “X” start-ABABABABABABABABABABABABAB-stop to be split into gene “Y” start-ABABABABABABA-stop and “Z” start-BABABABAB-stop as shown in (Fig. S1A) (Table S5, cluster 262). Here, a gene was identified in each of the eight Mtb strains examined. Seven of the eight strains had one copy of the gene, while in the eighth strain (BCCM093), the gene was split into two copies by fragmentation from a premature stop. Multiple paralogs within one strain and/or multiple orthologs among strains was a second cause of gene clusters. In the example shown (Fig. S1B) (Table S5, cluster 145), all strains except BCCM097 contained two paralogs of the canonical Rv3269. One of the paralogs present on each genome existed as either Rv3269-1 or Rv3269-2 which differed from each other by one nonsynonymous SNP. The second paralog present on each genome existed as Rv3269-3 or Rv3269-4 which also differed from each other by one nonsynonymous SNP. In the example of H37Rv, this strain contained "Rv3269-2" & "Rv3269-3" paralogs while BCCM101 contained "Rv3269-2" & "Rv3269-4" paralogs. BCCM097 only contained one of the two paralogs. Nearly identical gene duplications was a third cause of gene clusters. In the example shown (Fig. S1C) (Table S5, cluster 013), two strains had an additional seventh copy of the Anvi’o annotated IS285, whereas the remaining five strains had 6 copies present. Some copies were identical, others differed by only one nonsynonymous SNP and/or were of varying lengths, and other copies contained short regions that differed from this common sequence more substantially. In the case of BCCM083 and BCCM101 which had 7 copies, the additional sequence only differed by one SNP, causing us to attribute copy number variation to nearly identical gene duplication rather than the presence of paralogs as described in the previous cluster which had much more variability in sequences present. In an additional example of gene clusters caused by a combination of features. We also identified clusters containing varying copies of paralogs as well as nearly identical gene duplication. Finally, clusters could occur due to a combination of these previously described features. In the example shown (Fig. S1D) (Table S5, cluster 022) each Mtb strain contained multiple paralogs of the plc (phospholipase C) gene ( plcA , plcB and plcC ) in tandem. Anvi’o annotated all the open reading frames in this cluster as plc while other annotation programs named these paralogs separately as plcA , plcB and plcC , depending on the number of copies of the gene cluster present in the strain. Gene fragmentation further expanded the size of this gene cluster, where each plc paralog was either present as a canonical plc gene sequence or as one or more of the plc paralogs that were split due to a premature stop codon and secondary start codon causing the cluster to include between three to five copies per genome.). 19/25 (76%) of clusters were multi-copy solely due to gene fragmentation, an additional 3/25 (12%) were due to the presence of a novel paralogs in a subset of at least one strain, 1/25 (4%) was the result of a gene duplication in one strain, and 2/25 (8%) were due to a combination of these features. In addition to the findings in our random subset that gene fragmentation increased the size of gene clusters, we would also expect to find variations in gene sizes and potentially decreases in the size of gene clusters by gene fusions created through the loss of stop codons, such as the lineage 3 Rv2044c- pncA gene fusion that we noted in the accessory genome ( Fig. 4A-B ). Although annotation artifacts could also affect the size and number of gene clusters, we did not discover any annotation artifacts in our random sample of 25 clusters. Download figure Open in new tab Fig 4 Comparison of pncA among different strains of M. tuberculosis. (a) the pncA and its surrounding genes are highly conserved across strains. A frameshift deletion rv2044c resulted in its fusion with pncA in L3 strains (BCCM_091, BCCM_090, BCCM_089). Percent similarity is shown using the gradient scale in grey. (b) Evaluation of RNAseq data (accession: PRJEB9763) aligned with minimap demonstrated that canonical pncA is still expressed and at a higher level than any fusion protein. A lineage 3 mutation in pncA As described above, we noted that not all “new” accessory genes created by the loss of a stop codon were significantly expressed as fusion proteins. For example, a comparison of pncA and its surrounding genes revealed a highly conserved region ( Fig. 4A ), except where a frameshift deletion in the stop codon of rv2044c , a putative membrane protein, resulted in an out of frame gene fusion with pncA in all L3 strains [ 33 ]. Despite the apparent creation of this new large gene, L3 RNAseq data demonstrated that the canonical rv2044c and pncA were still separately expressed, each at high levels, compared to the new fusion RNA transcript, as can be seen by the low number of cDNA reads spanning the rv2044c - pncA junction compared to the cDNA reads aligning with the canonical open reading frame of ether gene ( Fig. 4B ). We assessed the functionality of PncA in strains with an rv2044c-pncA gene fusion by determining whether conversion of PZA to pyrazinoic acid (POA), the active form of PZA, was detectible in cultures of these strains. The two Lineage 3 strains with this gene fusions that we tested demonstrated sufficient PncA protein activity to fully hydrolyze PZA (Fig. S2). These results demonstrate that the effect of gene fusions on gene expression and bacterial phenotype should be confirmed with RNA and potentially protein or functional studies where warranted, but they also demonstrate how new genes can perhaps evolve in Mtb over time. Strain-specific accessory genes We found that each of the 50 strains contained an average of 1.78 genes that were present only in that strain ( Table 1 ). Annotation of these 89 “strain specific” accessory genes, or singletons, revealed that 29 were hypothetical genes, 3 were strain specific duplications and the remaining 37 were due to premature stop codons. All three identified duplications were found in BCCM085 and were duplications of an ATP dependent helicase, an NADH-ubiquinone oxidoreductase chain F and 3-phosphoshikimate 1-carboxyvinyltransferase respectively. The remaining accessory genes fell into the categories shown in Fig. 3D . Identifying hypervariable regions (HVRs) Using PPanGGOLiN, the 50 genomes were originally divided into 468 regions of genome plasticity (RGP) comprised of genomic islands and regions that were lost across all 50 genomes. Of the identified RGPs, 19 ‘spots’ were identified that that did not have the same gene content but had similar bordering elements. Given that we were evaluating several independent strains and not ‘plastic’ changes within one strain, we renamed RGP’s as hypervariable regions (HVRs). We also developed Table S6, which simplifies the identification of each HVR (i.e. HVR 1 – 19) by listing the names and H37Rv coordinates of the conserved genes which flank each HVR. Further investigation of these spots revealed that several of them were in tandem regions that we subsequently condensed into 16 HVRs. Clinker (v0.0.25, https://github.com/gamcil/clinker ) was used to visualize HVR organization [ 34 ]. Clinical strains depicted were randomly chosen as representative strains by PPanGGOLiN and commonly used reference stains (H37Rv, CDC1551, Erdman and HN878) were also included for comparison. HVR’s were broadly categorized into three groups: PE/PPE HVR’s, mobilome or transposon HVR’s, and a single lineage specific HVR. PE/PPE HVRs These HVRs were characterized by differences in PE/PPE genes contained within the HVR. PE/PPE HVR’s included: HVR2, HVR5, HVR6, HVR12, HVR15, and HVR16. HVR2, bordered by tasB and a hypothetical protein, was caused by the insertion of transposon and mobile elements throughout the region in multiple lineages. Interestingly, although RAST annotations indicated that PPE59 was missing in H37Rv and CDC1551, a BLAST search indicated that the sequence was present. The gene sequence for PPE57 and PPE58 in HVR2 however was missing in CDC1551 (Fig. S3B). HVR5, bordered by eccC5 and mycP5, varied by the presence of a conserved hypothetical gene in BCCM_092 (L4) and BCCM_102 (L1) (Fig. S3E). HVR6 was bordered by acrR and rv0281 (an O-methyltransferase). Variation in HVR6 was caused by the presence of a paralog of PE_PGRS3 in several strains, the deletion of PE_PGRS4 in HN878 and the presence of hypothetical genes scattered throughout the region in multiple lineages (Fig. S3F). HVR12 was bordered by plcA and glyS . Much of the variability in HVR12 was the result of random transposon insertions; however, we did observe duplications of esxN.2 and esxJ.3 and PPE38 in TDR116 and BCCM_3241. We also observed complete or partial disruptions of PPE38 and PPE39 as the result of transposon insertions (Fig. S3K). HVR15 was bordered by fadD15 and fadD19 . HVR15 demonstrated the variability observed across PE/PPE genes. We observed variable gene sizes of PE_PGRS55 , PE_PGRS56 and PE_PGRS57 . We also observed gene disruptions in PE_PGRS53 and PE_PGRS54 due to premature stop-codons and the presence of hypothetical genes found across multiple lineages ( Fig. 5C ). HVR16, bordered by PPE54 and rv3351c (a putative oxidoreductase), also showed variability in PE/PPE genes. We observed variable gene sizes of PPE54 as well as gene disruptions in PPE54 , PE_PGRS50 , and PPE56 due to premature stop codons and transposon elements that were also observed across multiple lineages (Fig. S3L). Download figure Open in new tab Fig 5 HVR regions across the M. tuberculosis Pangenome. (a-c) Schematics of (a) a mobilome or transposon HVR (b) a lineage HVR and (c) a PE/PPE HVR determined by PPanGGOLiN in representative strains and commonly used reference strains: H37Rv, CDC1551, Erdman, and HN878. Images were made using Clinker (57). Mobilome or transposon HVRs These HVRs all exhibited differences in mobilome, transposon, hypothetical or phage genes contained within the HVR. HVR’s in this group are: HVR1, HVR3, HVR4, HVR7, HVR8, HVR10, HVR11, HVR13, and HVR14. HVR1 was typified by the deletion of several phage related proteins and hypothetical proteins between border genes bioF1 and rv1591 in lineages 4 and 2 ( Fig. 5A ). The deletion was observed in CDC1551, Erdman and HN878 indicating that the deletion of this region was found in multiple lineages. HVR3, bordered by insertion sequence B9 and a hypothetical protein, was also found in multiple lineages and was caused by the deletion of several hypothetical genes in strains in Lineage 2 and 4, random transposon insertion and the deletion of mmpl14 , a probable transmembrane transport protein, in representative strain H37Rv (Fig. S3C). HVR4, bordered by a hypothetical protein and csm2 , was the result of random insertions of transposon elements, the deletion of several CRISPR associated proteins ( cas1 , csm6 , csm5 ) in HN878 and the insertion of several conserved hypothetical proteins in L1, L3, L5, L6 and L7 (Fig. S3D). HVR7, bordered by sdaA and cystathionine beta-lyase, was largely conserved except for the presence of small hypothetical genes in certain strains, the presence of a transposon element in BCCM_091 (L3) and the deletion of rv0071 (retron-type RNA-directed DNA polymerase), rv0072 (an ABC-type antimicrobial peptide transporter) and Xaa-Pro dipeptidase in HN878 (Fig. S3G). The deletion was not observed in other L2 strains. HVR8, bordered by fadD36 and rv1203c (conserved hypothetical protein), was also largely conserved except for small hypotheticals found in several strains in L4, 2, 3, 1 and 7 (Fig. S3H). HVR-10, bordered by moaX and dacB1 , like HVR2 and HVR3 was also largely the result of transposon and mobile elements throughout the region. Several strains across all lineages also contained several small conserved hypothetical genes. Representative strains H37Rv and Erdman also contained deletions or partial deletions of rv3324a (Pterin-4-alpha-carbinolamine dehydratase) and a GTP 3’,8-cyclase (Fig. S3I). HVR11, bordered by rv3015c (methionine synthase) and rv3026c (Acyl-CoA:1-acyl-sn-glycerol-3- phosphateacyltransferase), was also largely the result of transposon or mobile elements randomly inserted throughout the region. We also observed the presence of several conserved hypotheticals and the deletion of esxR and esxS in Erdman (L4), TDR116 (L4), BCCM_098 (L6) and all L3 strains (Fig. S3J). HVR13, bordered by rv1043c (a protease- related protein) and rv1049 (a transcriptional regulator) was largely conserved across strains, except for the presence of additional transposon elements in certain strains and the deletion of a hypothetical protein in CDC1551 (Fig. S3M). HVR14 was bordered by rv3463 (Coenzyme F420-dependent N5,N10-methylenetetrahydromethanopterin reductase and related flavin-dependent oxidoreductases) and ilvB2 . Several strains had a large insertion of several hypothetical or probable phage-related proteins that were observed across multiple lineages. The phage proteins were deleted in representative strains H37Rv (L4) and HN878 (L2) (Fig. S3A). Lineage HVR This category of HVR had lineage specific differences in gene content. HVR9, was bordered by rv1974 (conserved hypothetical) and mpt64 . Unlike the other HVRs, HVR9 exhibited differences that varied in the same way by all members of one or more specific lineages. Several hypothetical proteins present in L1, 2, 3, and 7 were deleted in lineage 4. All of the hypothetical genes in HVR9 in addition to rv1977 (a Zn-dependent protease with chaperone function), a large hypothetical gene and rv1979c (an uncharacterized amino acid permease) were deleted in L5 strains. In L6, the entire region was deleted except for mpt64 and rv1979c ( Fig. 5B ). The PGRR, a universal reference sequence for inter-strain studies We compiled a Pangenome Gene Reference Resource (PGRR) comprised of all genes and their paralogs to provide an efficient tool for performing isolated gene alignments and for accessing gene-level information from our pangenome reference strains. Paralogs were defined as any gene that had at least one SNP difference across the pangenome. One potential application of the PGRR is to summarize genomic variability for genes of interest across strains or lineages. For example, individual genome variability can be easily extracted, aligned, and compared across outcomes, such as drug tolerance or drug resistance. In addition, the PGRR can be used as a universal reference sequence to efficiently align DNA- or RNA-sequencing reads from non-reference sequences (such as those obtained from clinical strains) to identify strain-specific gene content or genome variability. To explore the use of the PGRR in this capacity, we first focused on the Mtb genome TbD1 region, a region that is found in Lineages 1, 5, 6, and 7 but which is absent in modern Mtb lineages. TbD1 includes mmpS6 and a longer 5’ region of mmpL6 . The PGRR contained 4 paralogs of mmpS6 . Each of these mmpS6 paralogs represented a different mmpS6 sequence variant from among all of the mmpS6 genes identified in our 50 sequenced Mtb strains (although in the case of mmpS6 , variants were only present in lineage 1, 5, 6 and 7). Similarly, the PGRR contained 5 paralogs of the longer mmpL6 gene. The PGRR also contained 4 paralogs of the shorter mmpL6 gene which is present in modern strains (e.g., Strains belonging to Lineage 2, 3 and 4). We identified 658 Illumina reads in our lineage 1 BCCM082 Mtb genome that mapped to TbD1. We then mapped these TbD1 reads to the lineage 4 H37Rv genome, where as expected only 45/658 (6.8%) of these reads aligned to the H37Rv since lineage 4 does not contain TbD1. We then mapped the 568 BCCM082 TbD1 reads to our PGRR, finding that all 658 (100%) of these reads aligned to the PGRR reference. All TbD1-mapped reads that aligned to the PGRR mapped to one or more of the 4 mmpS6 paralogs, or one or more of the 5 full length mmpL6 gene paralogs ( Fig. 6 ; Table S7). None of the TbD1 reads mapped to the truncated mmpL6 paralogs. Download figure Open in new tab Fig 6 Diagram Exploring the Utility of the PGRR with a Subset of Illumina Sequencing Reads. Illumina sequencing reads from Mtb. strain BCCM082’s genomic DNA were concatenated and aligned against BCCM082’s intact TbD1 sequence. Reads mapping to this region were then aligned to the PGRR using Bowtie 2 with the “very-sensitive-local” parameter. Identifiers formatted as “GN####” represent a paralog found within the PGRR. Corresponding RAST annotations and all representative strains containing the paralog’s exact sequence from ourstudy are shown. To evaluate the use of the PGRR to study sequencing data from whole strains, we next mapped all Illumina sequencing reads from strain BCCM082 to the complete genome of strain BCCM082 (i.e mapped to itself), H37Rv, and the PGRR using default Bowtie settings. We found that across these 2,129,345 reads, 43,964 (2.9%) aligned to the PGRR but not to H37Rv. Of the 43,964 BCCM082 reads that did not map to H37Rv, all aligned to one or more of the 17,357 paralogs of 4,294 genes present in the PGRR but not in H37Rv (Fig. S4). These genes included those with potential for drug targeting due to their roles in antimicrobial resistance, growth, and bacterial survival ( Table 3 ). These results demonstrate that the PGRR served its expected purpose as a pangenome mapping tool to accurately match new sequence reads with paralogs already identified in the pangenome including reads that might be otherwise missed or discarded when using the H37Rv genome as the sole reference. View this table: View inline View popup Table 3: PGRR average reads counts DISCUSSION Much of Mtb research, including drug and vaccine development work, is dependent on the use of a handful of reference strains. This approach is based on the assumption that the Mtb genome is highly conserved across the entire species. We show that although the Mtb pangenome is well conserved, there exist several regions of variability that may be clinically significant. By using highly accurate, closed out genomes representing all 7 established Mtb lineages, our approach enabled a comprehensive analysis of true strain-to-strain differences across a global cohort of strains. This work is supported by recent studies indicating that individual strains of Mtb are more diverse than previously thought [ 10 , 18 , 20 , 23 , 35 – 37 ]. Comparative genomic studies have also demonstrated differences in SNPs and INDELs, transposon elements and PE/PPE genes [ 23 , 35 , 38 , 39 ]. However, most whole genome based comparative analysis studies have predominantly used H37Rv as a reference suggesting that genes not found in H37Rv were likely discarded from analysis. These studies were also predicated on the idea that the H37Rv reference genome itself is correct which we have previously demonstrated is not the case [ 13 ]. Our analysis of Mtb core and accessory genes across 50 strains supported by highly accurate sequencing and de-novo assembly methods identified a pangenome that produces variable numbers of genes within a gene cluster both within a single genome or across different strains. Our identification of multi- copy core gene clusters suggests functional relevance, as gene duplications could be driven by increasing fitness. However, gene clusters caused by gene fragmentation may in fact suggest the opposite i.e. this class of core gene may not contribute to fitness in cases where neither gene fragment is biologically useful. Our observation that some gene clusters contain both intact duplications and intact closely related paralogs as well as fragmented paralogs and gene fusions suggests further evolutionary complexity where both paralog generation and gene fragmentation within a gene cluster could both provide a fitness benefit. We detected 16 HVR regions in our cohort. The majority of these HVRs involved variations in PE/PPE genes and gene neighborhoods. PE/PPE genes are highly repetitive and share significant homology across the entire family of genes. There are approximately 168 identified PE/PPE genes in Mtb [ 16 , 17 ]. Diversity across PE/PPE genes is mostly found in the C-terminal domain of the protein. Despite the observed diversity in PE/PPE genes, we are not aware of any studies to date that have evaluated strain or lineage specific changes in PE/PPE genes or differences in sequence homology across conserved PE/PPE genes. We observed that 6/16 HVRs involved variability either in PE/PPE genes or in PE/PPE gene neighborhoods. We and others have previously described HVR-12, but a comparison across our pangenome strains revealed even more diversity than was previously known [ 13 , 40 ]. We also identified duplications of PPE38, esxN.2 and esxJ.3 in TDR116 (L4) and BCCM_3241 (L2) ( Fig. 5l ). PPE38 is required for the secretion of all PE_PGRS and PPE-MPTR proteins and mutations have been linked to increased virulence [ 41 , 42 ]. It is likely that there are additional differences in PE/PPE genes outside of our HVRs that will require further analysis and validation. Six of the HVRs were the result of random transposon or mobile element insertions throughout the region. Although HVR-5 and HVR-13 were technically identified as HVRs by PPanGGOLiN, the variability in these regions were due to small annotated hypothetical genes in certain strains. Two HVRs, HVR-1 and HVR-14, included large probable prophage phiRv protein insertions in several strains. Full-length prophages are comprised of multiple components, including complete excision and integration cassette, lysis cassette, and bacteriophage structure proteins. Although the presence of variable numbers of phiRv have been previously described, their exact function in Mtb is still a mystery [ 17 , 43 ]. In addition to identifying novel genes and hypervariable regions, we were able to aggregate the intergenic diversity observed across our pangenome into a single resource which we have called PGRR. This resource enables the rapid retrieval of gene sequences from a single, multiple, or all strains in the pangenome. This resource will facilitate multiple types of sequence base analyses, including the analysis of gene-level sequence variation across strains within/across gene isoforms or strains, multi-sequence alignment and phylogenetic analysis of coding sequences of TB genes, and the accurate alignment of DNA or RNA sequencing data from clinical strains. For the latter specifically, the PGRR resource will directly align sequences to tens of thousands of gene isoforms, assuring the most accurate alignment and determination of gene content within the sequenced TB strain. This application is most relevant in cases where we do not have the reference genome at hand or when the genome is not annotated. The PGRR not only enhances alignment accuracy but also accommodates the variant nature of the Mtb pangenome, reducing or eliminating false alignments and preventing sequence reads from being eliminated entirely. This tool is versatile and expansible in that it facilitates the convenient incorporation of additional genomes as they are sequenced in future amendments. Compared to prior Mtb comparative pangenome studies [ 44 ] [ 45 ], our approach differs in several key methodological and analytical aspects. Previous pangenome analyses have often relied on large numbers of draft assemblies generated primarily from short-read sequencing, with gene presence–absence patterns inferred relative to a reference genome such as H37Rv. Other recent work has combined long- and short-read assemblies for a smaller set of isolates to investigate broad-scale structural variation, but without a formal or systematic identification of recurrent hypervariable regions (HVRs) many within or adjacent to PE/PPE genes, which we found to account for substantial diversity between strains both within and between lineages. Our strategy also enabled reference-independent calculation of lineage-specific SNP and INDEL counts, revealing patterns that differ from strictly reference-based methods, and allowed the discovery of previously undescribed duplications in virulence-associated genes such as PPE38, esxN.2, and esxJ.3. Although not comprehensive, our manual curation of a subset of core gene clusters enabled us to identify genetic drivers of gene cluster development. These findings, especially if further expanded, should provide insights into how the Mtb pangenome generates diversity in the absence of horizontal gene transfer. Furthermore, our publicly available PGRR resource provides a flexible platform for retrieving lineage- and strain-specific sequences, enabling downstream applications such as targeted variant analysis and improved clinical sequence alignment—capabilities not offered by earlier studies. These methodological and analytical advances strengthen the accuracy, resolution, and translational potential of our findings, providing a more comprehensive and reliable framework for understanding strain-specific diversity in Mtb. While this study provides a robust and comprehensive pangenome analysis of Mtb, several limitations should be noted. Firstly, our analysis is based on 50 complete genomes, which, although diverse, may not fully capture the entire genetic diversity present in the species, particularly in underrepresented or newly discovered lineages. Additionally, while we have identified a number of hypervariable regions (HVRs), the functional relevance of these regions, particularly in relation to pathogenicity or clinical outcomes, remains an area for further investigation. Final, while our PGRR resource is a valuable tool for sequence alignment, it is limited by the availability of high-quality, annotated genome data. Continued sequencing efforts and the development of additional strain-specific resources will be essential for further refining this database and enhancing its utility in clinical and research settings. DATA AVAILABILITY Whole genome sequencing data for the strains used in this study have been deposited in the NCBI sequence read archive under SRA Project number PRJNA1010209. Complete, annotated genomes for the strains in this study have been deposited in the NCBI GenBank under project number PRJNA1010209. Supplementary tables referenced in this study have been deposited to https://doi.org/10.5281/zenodo.17496644 . Pangenomic analysis data from Anvio used in this study have been deposited to https://github.com/wejlab/PGRR-supplementary . The Pangenome Gene Reference Resource sequence data have been deposited to https://github.com/wejlab/PGRR . CODE AVAILABILITY Zenodo: Underlying data for ‘Defining the Mycobacterium tuberculosis Pangenome and Suggestions for a New Composite Reference Sequence.’. https://doi.org/10.5281/zenodo.10426196 CONTRIBUTIONS P.C., A.R.O., W.E.J., and D.A. designed the study; P.C., E.M.O., and C.A.G. library prepped and sequenced the samples; P.C., A.R.O., A.D.L., E.M.O., P.K., K.A.V. and E.C.F performed data analysis; P.C.,A.R.O., E.M.O., K.A.V., P.K., H.F., A.H., and A.M.E. contributed to data visualization. P.C., E.M.O., H.F., A.H., A.M.E., and D.A. wrote the manuscript. W.E.J., K.A.V., S.D.M. and P.K. edited the manuscript. ACKNOWLDEGEMENTS Research reported in this publication was supported by the National Institute of Allergy and Infectious Diseases of the National Institutes of Health under award numbers U19AI11276 and U19AI162598 and Aubrey R. Odom was funded by T32HL125232. The content is solely the responsibility of the authors and does not necessarily represent the official views of the National Institutes of Health. The authors acknowledge the Genomics Center Rutgers New Jersey Medical School ( https://research.njms.rutgers.edu/genomics/ ) and the Office of Advanced Research Computing (OARC) at Rutgers, The State University of New Jersey ( http://oarc.rutgers.edu ) for providing access to the Amarel cluster and associated research computing resources that have contributed to the results reported here. Funder Information Declared National Institute of Allergy and Infectious Diseases, https://ror.org/043z4tv69 , U19AI11276 , U19AI162598 , T32HL125232 REFERENCES 1. ↵ Global Tuberculosis Report 2022 . 2022 . 2. ↵ Marcel , A.B. and G. Sébastien , The Rise and Fall of the Mycobacterium tuberculosis Complex . Genetics and Evolution of Infectious Diseases , 2011 : p. 651-667, publisher = Elsevier Inc . 3. ↵ Dick Van , S. , et al. , A novel pathogenic taxon of the Mycobacterium tuberculosis complex, Canetti: characterization of an exceptional isolate from Africa . International journal of systematic bacteriology , 1997 . 47 : p. 1236 – 1245 , pmid = 9336935, publisher = Int J Syst Bacteriol . OpenUrl CrossRef PubMed 4. ↵ Mohammad , J.N. , et al. , New insights in to the intrinsic and acquired drug resistance mechanisms in mycobacteria , in Frontiers in Microbiology . 2017 . p. 681, publisher = Frontiers Research Foundation . 5. Mashael , A.-S. and A.-H. Sahal , Diversity and evolution of drug resistance mechanisms in Mycobacterium tuberculosis , in Infection and Drug Resistance . 2017 . p. 333-342, publisher = Dove Medical Press Ltd . 6. Maha , R.F. , et al. , Genetic determinants of drug resistance in mycobacterium tuberculosis and their diagnostic value . American Journal of Respiratory and Critical Care Medicine , 2016 . 194 : p. 621 – 630 , pmid = 26910495, publisher = American Thoracic Society . OpenUrl CrossRef PubMed 7. Timothy , M.W. , et al. , Whole-genome sequencing for prediction of Mycobacterium tuberculosis drug susceptibility and resistance: a retrospective cohort study . 2015 . 8. ↵ Mireia , C. and G. Sebastien , Consequences of genomic diversity in mycobacterium tuberculosis , in Seminars in Immunology . 2014 . p. 431-444, publisher = Academic Press . 9. Juan Carlos , P. and M. Anandi , Drug resistance mechanisms in Mycobacterium tuberculosis , in Antibiotics . 2014 . p. 317-340, publisher = MDPI AG . 10. ↵ Andrej , T. , et al. , Evolution of drug resistance in tuberculosis: Recent progress and implications for diagnosis and therapy . Drugs , 2014 . 74 : p. 1063-1072, publisher = Springer International Publishing . 11. Tasha , S. , A.W. Kerstin , and N. Liem , Molecular Biology of Drug Resistance in Mycobacterium tuberculosis . Current topics in microbiology and immunology . Vol. 374. 2012 . 53-80, pmid = 23179675, publisher = NIH Public Access . 12. ↵ Iseman , M.D. , Tuberculosis therapy: past, present and future . European Respiratory Journal , 2002 . 20 : p. 87S-94s, pmid = 12168751, publisher = European Respiratory Society . 13. ↵ Poonam , C. , et al. , A comprehensive update to the Mycobacterium tuberculosis H37Rv reference genome . Nature Communications 2022 13:1, 2022. 13 : p. 1-12, publisher = Nature Publishing Group . 14. Mickael , O. , et al. , Pathogenomic analyses of Mycobacterium microti, an ESX-1-deleted member of the Mycobacterium tuberculosis complex causing disease in various hosts . Microbial genomics , 2021 . 7 : p. 1-18, pmid = 33529148, publisher = Microb Genom . 15. ↵ Louis , S.A. , et al. , Mutations in ppe38 block PE_PGRS secretion and increase virulence of Mycobacterium tuberculosis . Nature Microbiology 2018 3:2, 2018. 3 : p. 181-188, pmid = 29335553, publisher = Nature Publishing Group . 16. ↵ Ates , L.S ., New insights into the mycobacterial PE and PPE proteins provide a framework for future research . Mol Microbiol , 2020 . 113 ( 1 ): p. 4 – 21 . OpenUrl CrossRef PubMed 17. ↵ Cole , S.T. , et al. , Deciphering the biology of Mycobacterium tuberculosis from the complete genome sequence . Nature , 1998 . 393(6685): p. 537-44. 18. ↵ Vinita , P. , et al. , Comparative Whole-Genome Analysis of Clinical Isolates Reveals Characteristic Architecture of Mycobacterium tuberculosis Pangenome . Plos One , 2015 . 10 : p. e0122979 . OpenUrl CrossRef PubMed 19. Erol , S.K. , et al. , Machine learning and structural analysis of Mycobacterium tuberculosis pan-genome identifies genetic signatures of antibiotic resistance . Nature Communications , 2018 . 9 . 20. ↵ Syed Beenish , R. , A.O. Egon , and S. Sarman , Pan-genome analysis of mycobacterium tuberculosis identifies accessory genome sequences deleted in modern Beijing lineage ., in bioRxiv. 2020. p. 91-755, publisher = bioRxiv. 21. ↵ Hamza Arshad , D. , et al. , Pangenome analysis of mycobacterium tuberculosis reveals core-drug targets and screening of promising lead compounds for drug discovery . Antibiotics , 2020 . 9 : p. 1-14, publisher = MDPI AG . 22. ↵ Vincent , V. , et al. , The TDR Tuberculosis Strain Bank: A resource for basic science, tool development and diagnostic services , in International Journal of Tuberculosis and Lung Disease . 2012 . p. 24-31. 23. ↵ Sònia , B. , et al. , Reference set of Mycobacterium tuberculosis clinical strains: A tool for research and product development . PloS one , 2019 . 14 . 24. ↵ Martin , H. , et al. , Antibiotic resistance prediction for Mycobacterium tuberculosis from genome sequence data with mykrobe version 1; peer review: 2 approved, 1 approved with reservations . Wellcome Open Research , 2019 . 4 : p. 191, publisher = F1000 Research Ltd . 25. ↵ Jérôme , A. , et al. , Genomics and Machine Learning for Taxonomy Consensus: The Mycobacterium tuberculosis Complex Paradigm . Plos One , 2015 . 10 : p. e0130912, pmid = 26154264, publisher = Public Library of Science . 26. ↵ Harrison , L.B. , V. Kapur , and M.A. Behr , An imputed ancestral reference genome for the Mycobacterium tuberculosis complex better captures structural genomic diversity for reference-based alignment workflows . Microb Genom , 2024 . 10 ( 1 ). 27. ↵ Daniela , B. and G. Sebastien , the nature and evolution of genomic diversity in the mycobacterium tuberculosis Complex . Advances in Experimental Medicine and Biology , 2017 . 1019: p. 1-26, pmid = 29116627, publisher = Springer New York LLC . 28. ↵ Hervé , T. , et al. , Genome analysis of multiple pathogenic isolates of Streptococcus agalactiae: Implications for the microbial ’’pan-genome’’ . 2005 . 29. George , V. , et al. , Ten years of pan-genome analyses, in Current Opinion in Microbiology . 2015 . p. 148-154, publisher = Elsevier Ltd . 30. ↵ James , O.M. , M. Alan , and J.O.C. Mary , Why prokaryotes have pangenomes , in Nature Microbiology . 2017 . p. 1-5, pmid = 28350002, publisher = Nature Publishing Group . 31. ↵ Jessica , C. , O. Isabel , and S. Sofía , In-depth Analysis of IS6110 Genomic Variability in the Mycobacterium tuberculosis Complex . Frontiers in Microbiology , 2022 . 13 : p. 512, publisher = Frontiers Media S.A. 32. ↵ Tanmoy , R. , M. Saurav , and B. Alok , Analysis of IS6110 insertion sites provide a glimpse into genome evolution of Mycobacterium tuberculosis . Scientific Reports 2015 5 : 1 , 2015 . 5 : p. 1-10, pmid = 26215170, publisher = Nature Publishing Group . OpenUrl 33. ↵ Ramani , B. , et al. , Analysis of mutations in pncA reveals non-overlapping patterns among various lineages of Mycobacterium tuberculosis . Scientific Reports 2018 8 : 1 , 2018 . 8 : p. 1-9, pmid = 29545614, publisher = Nature Publishing Group . OpenUrl PubMed 34. ↵ Cameron , L.M.G. and C. Yit Heng , Clinker & clustermap.js: Automatic generation of gene cluster comparison figures . Bioinformatics (Oxford, England), 2021 . 37 : p. 2473-2475, pmid = 33459763, publisher = Bioinformatics . 35. ↵ Brosch , R. , et al. , A new evolutionary scenario for the Mycobacterium tuberculosis complex . Proceedings of the National Academy of Sciences of the United States of America , 2002 . 99 : p. 3684, pmid = 11891304, publisher = National Academy of Sciences . 36. Sebastien , G. , Ecology and evolution of Mycobacterium tuberculosis . Nature Reviews Microbiology 2018 16:4, 2018. 16 : p. 202-213, pmid = 29456241, publisher = Nature Publishing Group . 37. ↵ Tingting , Y. , et al. , Pan-Genomic Study of Mycobacterium tuberculosis Reflecting the Primary/Secondary Genes, Generality/Individuality, and the Interconversion Through Copy Number Variations . Frontiers in Microbiology , 2018 . 9 : p. 1886, publisher = Frontiers Media S.A . 38. ↵ Christopher , R.E.M. , et al. , Comparative Analysis of Mycobacterium tuberculosis pe and ppe Genes Reveals High Sequence Variation and an Apparent Absence of Selective Constraints . Plos One , 2012 . 7 : p. e30593, pmid = 22496726, publisher = Public Library of Science . 39. ↵ Mireia , C. , et al. , Phylogenomics of Mycobacterium africanum reveals a new lineage and a complex evolutionary history . Microbial Genomics , 2021 . 7 : p. 477 . OpenUrl 40. ↵ Christopher Re , M. , et al. , Evidence for a rapid rate of molecular evolution at the hypervariable and immunogenic Mycobacterium tuberculosis PPE38 gene region . BMC Evolutionary Biology , 2009 . 9 : p. 237, pmid = 19769792, publisher = BioMed Central . 41. ↵ Louis , S.A. , New insights into the mycobacterial PE and PPE proteins provide a framework for future research . Molecular Microbiology , 2020 . 113 : p. 4-21, pmid = 31661176, publisher = John Wiley & Sons, Ltd . 42. ↵ Louis , S.A. , et al. , RD5-mediated lack of PE_PGRS and PPE-MPTR export in BCG vaccine strains results in strong reduction of antigenic repertoire but little impact on protection . 2018 . 43. ↵ Xiangyu , F. , A. Abu Algasim Elgaili Abd , and X. Jianping , Distribution and function of prophage phiRv1 and phiRv2 among Mycobacterium tuberculosis complex . Journal of Biomolecular Structure and Dynamics , 2016 . 34 : p. 233-238, pmid = 25855385, publisher = Taylor and Francis Ltd . 44. ↵ Behruznia , M. , et al. , The Mycobacterium tuberculosis complex pangenome is small and driven by sub-lineage-specific regions of difference . eLife , 2024 . 13 . 45. ↵ Marin , M.G. , et al. , Analysis of the limited M. tuberculosis accessory genome reveals potential pitfalls of pan-genome analysis approaches . bioRxiv , 2024 . View the discussion thread. Back to top Previous Next Posted November 03, 2025. Download PDF Supplementary Material Email Thank you for your interest in spreading the word about bioRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following Defining the Mycobacterium tuberculosis Pangenome and Suggestions for a New Composite Reference Sequence Message Subject (Your Name) has forwarded a page to you from bioRxiv Message Body (Your Name) thought you would like to see this page from the bioRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share Defining the Mycobacterium tuberculosis Pangenome and Suggestions for a New Composite Reference Sequence Poonam Chitale , Elissa Ocke , Aubrey R. Odom , Howard Fan , Alexander Henoch , Karla Vasco , Emily C. Fogarty , Courtney Grady , Alexander D. Lemenze , Pradeep Kumar , Shannon Manning , A. Murat Eren , W. Evan Johnson , David Alland bioRxiv 2025.11.03.686283; doi: https://doi.org/10.1101/2025.11.03.686283 Share This Article: Copy Citation Tools Defining the Mycobacterium tuberculosis Pangenome and Suggestions for a New Composite Reference Sequence Poonam Chitale , Elissa Ocke , Aubrey R. Odom , Howard Fan , Alexander Henoch , Karla Vasco , Emily C. Fogarty , Courtney Grady , Alexander D. Lemenze , Pradeep Kumar , Shannon Manning , A. Murat Eren , W. Evan Johnson , David Alland bioRxiv 2025.11.03.686283; doi: https://doi.org/10.1101/2025.11.03.686283 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Microbiology Subject Areas All Articles Animal Behavior and Cognition (7629) Biochemistry (17660) Bioengineering (13881) Bioinformatics (41910) Biophysics (21436) Cancer Biology (18576) Cell Biology (25480) Clinical Trials (138) Developmental Biology (13368) Ecology (19887) Epidemiology (2067) Evolutionary Biology (24302) Genetics (15598) Genomics (22482) Immunology (17726) Microbiology (40360) Molecular Biology (17163) Neuroscience (88534) Paleontology (666) Pathology (2830) Pharmacology and Toxicology (4821) Physiology (7637) Plant Biology (15129) Scientific Communication and Education (2045) Synthetic Biology (4290) Systems Biology (9817) Zoology (2269)
Text is read by the "Ask this paper" AI Q&A widget below.
Extraction quality varies by source — PMC NXML preserves structure
cleanly, OA-HTML may include some navigation residue, and OA-PDF can
have broken hyphenation. The publisher copy
(via DOI)
is the canonical version.