GeneScanner: profiling genetic variation across bacterial populations

preprint OA: gold CC-BY-4.0
📄 Open PDF Full text JSON View at publisher

Abstract

Rapid, low-cost genome sequencing has transformed microbiology, advancing efforts to link genetic and phenotypic variation. In laboratory settings, genome-wide functional screens of reference strains are revealing genes and mechanisms underlying important phenotypes. Simultaneously, population-scale comparative genomics describes the breadth of natural genetic and phenotypic diversity across lineages. These approaches provide complementary but contrasting information. Whereas laboratory studies offer causal understanding within simplified systems, population-level analyses capture ecological realism but are largely limited to detecting associations rather than cause-and-effect relationships. Linking laboratory-derived mutations to natural population variation remains challenging, particularly for researchers lacking bioinformatics expertise. Here, we present GeneScanner, a user-friendly tool that facilitates analysis of gene- and protein-level variation across large bacterial genome collections. GeneScanner detects genetic variation and amino acid substitutions in homologous sequence and supports genotype–phenotype association studies. By bridging laboratory functional genomics data with genomic diversity across natural populations, GeneScanner enhances functional interpretation of microbial variation in the real world, supporting research in antimicrobial resistance, virulence, and pathogen surveillance.
Full text 67,288 characters · extracted from preprint-html · click to expand
GeneScanner: profiling genetic variation across bacterial populations | bioRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-M677548'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search New Results GeneScanner: profiling genetic variation across bacterial populations View ORCID Profile Carolin M. Kobras , View ORCID Profile Seungwon Ko , View ORCID Profile Priyanshu S. Raikwar , View ORCID Profile Broncio Aguilar-Sanjuan , View ORCID Profile Keith A. Jolley , View ORCID Profile Samuel K. Sheppard doi: https://doi.org/10.1101/2025.11.02.685864 Carolin M. Kobras 1 Sir William Dunn School of Pathology, University of Oxford , Oxford, United Kingdom Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Carolin M. Kobras Seungwon Ko 2 Ineos Oxford Institute for Antimicrobial Research, Department of Biology, University of Oxford , Oxford, United Kingdom Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Seungwon Ko Priyanshu S. Raikwar 2 Ineos Oxford Institute for Antimicrobial Research, Department of Biology, University of Oxford , Oxford, United Kingdom Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Priyanshu S. Raikwar Broncio Aguilar-Sanjuan 2 Ineos Oxford Institute for Antimicrobial Research, Department of Biology, University of Oxford , Oxford, United Kingdom Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Broncio Aguilar-Sanjuan Keith A. Jolley 2 Ineos Oxford Institute for Antimicrobial Research, Department of Biology, University of Oxford , Oxford, United Kingdom Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Keith A. Jolley Samuel K. Sheppard 2 Ineos Oxford Institute for Antimicrobial Research, Department of Biology, University of Oxford , Oxford, United Kingdom Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Samuel K. Sheppard For correspondence: Samuel.Sheppard{at}biology.ox.ac.uk Abstract Full Text Info/History Metrics Supplementary material Data/Code Preview PDF Abstract Rapid, low-cost genome sequencing has transformed microbiology, advancing efforts to link genetic and phenotypic variation. In laboratory settings, genome-wide functional screens of reference strains are revealing genes and mechanisms underlying important phenotypes. Simultaneously, population-scale comparative genomics describes the breadth of natural genetic and phenotypic diversity across lineages. These approaches provide complementary but contrasting information. Whereas laboratory studies offer causal understanding within simplified systems, population-level analyses capture ecological realism but are largely limited to detecting associations rather than cause-and-effect relationships. Linking laboratory-derived mutations to natural population variation remains challenging, particularly for researchers lacking bioinformatics expertise. Here, we present GeneScanner, a user-friendly tool that facilitates analysis of gene- and protein-level variation across large bacterial genome collections. GeneScanner detects genetic variation and amino acid substitutions in homologous sequence and supports genotype–phenotype association studies. By bridging laboratory functional genomics data with genomic diversity across natural populations, GeneScanner enhances functional interpretation of microbial variation in the real world, supporting research in antimicrobial resistance, virulence, and pathogen surveillance. Background For decades, molecular microbiology has sought to understand the genetic basis of important bacterial traits including pathogenicity, antimicrobial resistance, and host interactions. Advances in sequencing technologies have transformed understanding of gene function, enabling the genome-wide analyses that define the field of functional genomics. For example, the principals of gene inactivation studies that compare the phenotypes of mutant and wild-type strains are now often augmented with genome-wide gene deletion [ 1 – 4 ] and transposon-insertion libraries [ 5 – 10 ], allowing rapid genotype-phenotypic screening under diverse selective conditions. Furthermore, in laboratory evolution experiments, where bacteria are subjected to selective pressures such as antibiotics or environmental stress [ 11 – 14 ], whole genome sequencing (WGS) is often used to compare progenitor and adapted isolates. While scientifically rigorous, these laboratory studies are often performed using a single laboratory adapted strain and may be limited in their ecological relevance. These approaches emphasize the importance of controlled experimental conditions and precise understanding of the ancestral and derived genetic background of laboratory strains, enabling targeted investigation of genotype-phenotype association. However, a laboratory settings may not fully reflect the complexity of natural environments where bacteria evolve [ 15 ]. The impact of WGS in a laboratory setting is mirrored in natural populations. Here the emphasis is often upon multi-strain comparative genomics for pathogen surveillance, phylogenetics and epidemiology. Millions of bacterial sequences are now freely available in public repositories [ 16 , 17 ] and curated databases [ 18 – 20 ], with metadata such as isolation source, place and time. This has transformed understanding of microbial genomics in the ‘wild’ with approaches including genome-wide association studies (GWAS) and covariation analyses revealing genetic variation underlying phenotypes linked to host adaptation [ 21 – 24 ], biofilm and virulence [ 25 – 30 ], and antibiotic resistance [ 31 – 34 ]. However, while some of the limitations of laboratory studies are overcome by analysing multiple strains in natural populations, highly relevant population genomics studies, lack the control and precision of laboratory functional genomics. This typically limits genotype-phenotype inference to association rather than cause-and-effect relationships. Next-generation microbiology brings the rigour of the laboratory together with the relevance of population-wide studies and has major potential for understanding gene function in natural systems [ 35 ]. For example, studies of bacterial pathogen including Campylobacter [ 23 ] and Streptococcus pneumoniae [ 36 ] confirm laboratory observations in natural isolate populations allowing accurate analysis of functional genomics at a population scale. While database resources and isolate genome collections are freely available for comparable analyses of other pathogen species, analysis pipelines are often aimed at bioinformaticians rather than laboratory microbiologists. This can require the installation of specific operating systems and environments and command-line experience for single nucleotide polymorphism (SNP) calling [ 37 – 41 ]. To address this challenge, we developed GeneScanner, a user-friendly tool that enables researchers to explore genetic variation in specific genes or proteins across large datasets from natural bacterial populations. Implemented as both a stand-alone script on GitHub and a web-plugin available through PubMLST [ 20 ], GeneScanner simplifies the comparison of genetic features across strains and the identification of variants linked to phenotypic traits such as antibiotic resistance or pathogenicity. By providing an intuitive interface, it allows users without extensive bioinformatics expertise to analyse large datasets, contextualizing laboratory-observed genotypes within real-world bacterial diversity. In this study, we analyse synthetic data and three case studies across different bacterial species and phenotypes. We demonstrate how GeneScanner detects nucleotide and protein-level variation associated with specific traits, underscoring its broad applicability in microbial genomics, from antibiotic resistance studies to pathogen surveillance and laboratory evolution experiments. Results GeneScanner is a user-friendly tool for exploring genetic variation in bacterial populations GeneScanner is a Python-based workflow, available on Github ( https://github.com/Sheppard-Lab/GeneScanner ), that accepts pre-aligned nucleotide or protein FASTA files. It screens every alignment column for sequence variation and collates the results into a multi-sheet workbook in .xlsx format. Even without bioinformatics background, GeneScanner can be used as a plugin on PubMLST ( https://bigsdb.readthedocs.io/en/latest/data_analysis/genescanner.html ) [ 20 ], supporting isolate selection and input file generation directly from organism-specific databases, and eliminating the need for local installation or command-line operation. GeneScanner operates in four sequential stages ( Fig. 1 ). In stage 1, the input alignment undergoes rigorous coverage and identity filtering, ensuring that only sequences meeting the predefined thresholds for ungapped alignment coverage and sequence identity to the reference are retained for subsequent analysis. In stage 2, the filtered sequences are analysed according to the user-selected mode (nucleotide or protein). In nucleotide mode, GeneScanner traverses the alignment both position-by-position and in codon triplets. Assuming no insertion or deletion events, each translated codon is compared with the corresponding reference codon to identify synonymous and non-synonymous substitutions, while insertions, deletions, and premature stop codons are recorded separately. In protein mode, the algorithm examines each aligned amino-acid column to quantify substitutions, gaps, and stop codons, and to calculate both mutation frequency and the remaining number of sequences following the stop codons. When the user requests protein analysis using nucleotide input, it automatically translates and realigns the sequences before proceeding as if native protein data had been provided. Download figure Open in new tab Figure 1. The GeneScanner workflow consists of four distinct steps. Schematic illustration of the GeneScanner workflow: (1) input parsing and quality filtering of alignment files (input); (2) codon-aware nucleotide or residue-aware protein variant calling; (3) optional per-group reanalysis; (4) output: xlsx format report generation with analysis, matrix, and summary worksheets. External tools MAFFT and snp-sites are invoked when translation, realignment or VCF output are requested. In stage 3, GeneScanner can optionally perform grouping analyses when the user defines categories for sequences. This categorization helps subdivide the population, for example by phenotype or other defining traits. The program partitions the sequence dataset according to these categories and re-runs the full analytical pipeline for each subset. In stage 4, GeneScanner generates comprehensive nucleotide and amino acid mutation analysis spreadsheets that include detailed mutation categories. It also produces sparse mutation matrices in both nucleotide and protein modes, in which rows represent isolates, columns correspond to alignment positions, and each filled cell contains the variant symbol when it differs from the reference. Each matrix is written to a dedicated worksheet to facilitate downstream visualization and inspection. GeneScanner also calculates and exports summary statistics to a dedicated worksheet. For nucleotide analyses, the summary includes the total alignment length, the number of mutated sites, positions with mutation frequencies exceeding 20%, and cumulative counts of synonymous and non-synonymous substitutions, and insertions, deletions, and stop-codons. For protein analyses, the corresponding statistics include substitutions, insertions, deletions, stop codons, and position-wise mutation frequencies. When multiple isolate groups are defined, GeneScanner generates separate worksheets and mutation matrices for each group, ensuring consistent column definitions and analytical parameters across all outputs. If distinct reference sequences are specified for individual groups, the program automatically reassigns the appropriate reference before performing variant calling. Benchmarking GeneScanner using a controlled synthetic dataset To demonstrate the functionality of GeneScanner, we first tested it on a controlled synthetic dataset of 100 300bp sequences in which the locations, types, and frequencies of mutations were explicitly designed (Supplementary file 1). The dataset reproduced the intended patterns ( Fig. 2A+B ), and GeneScanner successfully captured variation at both nucleotide and protein levels (Supplementary File 2). At the nucleotide level, 20% and 30% synonymous variants occurred at positions 12 and 147, respectively, while 20% and 30% nonsynonymous variants appeared at positions 46 and 181, alongside a 10% nonsense mutation at position 210. When translated, synonymous sites produced no residue changes, whereas nonsynonymous mutations manifested as 20% Q16E and 30% S61P substitutions. The stop-gain mutation was correctly assigned to amino acid position 70, with GeneScanner distinguishing it from other nonsynonymous substitutions. Download figure Open in new tab Figure 2. GeneScanner can reliably detect sequence variation on a nucleotide and protein level. Analysis of a synthetic dataset demonstrates and validates the functionality of GeneScanner. The tool successfully captures synonymous and non-synonymous mutations including those leading to stop codons at both nucleotide (A) and protein (B) levels. Insertions and deletions are recognised as events by nucleotide analysis (C), and as frameshifts (FS), premature stop codons or frame-restoration, where applicable, at the protein level (D). The correct detection of insertion and deletion (indel) events further highlighted the flexibility of the workflow ( Fig. 2C+D , Supplementary files 3+4). A single-nucleotide deletion at position 4 produced a frameshift with an early stop at amino acid position 14 (10%), and a single-nucleotide insertion at position 69 generated a frameshifted protein truncated at position 29 (10%). In the frame-restoration scenario, a paired insertion at position 153 (5%) and deletion at position 219 (5%) created a transient frameshift with amino acid substitutions between residues 52–73, after which the original reading frame was recovered. These results illustrate how GeneScanner’s dual nucleotide-protein mode can trace both the disruption and recovery of coding frames, preserving alignment quality while recording complex mutational events. GeneScanner validation of in silico sequence evolution across two populations To further validate the functionality of GeneScanner on datasets with unknown outcomes, and to test its grouping feature for analysing distinct subpopulations, we simulated evolution with differential selection. We created two populations of 5,000 bacterial genomes, starting with a random 300 bp coding sequence. Mutations occurring in a specific region (positions 100–130) were set to have opposite fitness effects in the two populations, while mutations elsewhere were neutral. After 1,000 generations, we sampled 500 isolates from each population (p1 and p2) and ran GeneScanner ( Fig. 3A , Supplementary files 5+6). The imposed selection produced clear divergence within the targeted 100-130 bp window ( Fig. 3B+C ). Using the original sequence before simulation as a reference, subpopulation p2 exhibited minor variation across the locus, consistent with neutral dynamics outside the selected window. In contrast, subpopulation p1 displayed a concentration of both synonymous and nonsynonymous differences restricted to positions 100-130, reflecting the impact of opposing selection pressures. Outside of the region under simulated, both p1 and p2 showed similar mutational patterns. GeneScanner successfully recapitulated these differences at nucleotide levels, confirming the tracking of mutational events across divergent evolutionary trajectories. Download figure Open in new tab Figure 3. GeneScanner reliably detects sequence variation from in vitro sequence evolution under differential selection. A: Simulated sequence evolution produced divergent outcomes between two subpopulations exposed to opposing selection pressures. B+C: Using the original reference sequence, GeneScanner detected only scattered minor variation across the locus in p2, whereas p1 exhibited concentrated synonymous and nonsynonymous substitutions within the 100-130 bp window. These differences captured for both subpopulations, demonstrating the tool’s ability to resolve selection-driven divergence. Together, these analyses demonstrate the utility of GeneScanner for detecting, classifying, and contextualizing intra-genic variation in bacterial populations. Using a controlled synthetic dataset, the workflow reproduced expected mutation patterns, and in simulations of in silico sequence evolution across different subpopulations, it accurately traced selection-driven divergence. This proof-of-concept evaluation established the four-stage GeneScanner pipeline - from alignment parsing to mutation classification and summary statistic generation. Building on this foundation, GeneScanner was used in three case studies spanning distinct bacterial species, phenotypes, and research contexts. Case study 1: Genetic variation of DNA replication genes is closely linked to ciprofloxacin resistance Antibiotic resistance is a major global health concern, with Escherichia coli being among the most problematic pathogens [ 42 ]. The fluoroquinolone antibiotic ciprofloxacin is a commonly used treatment that targets essential enzymes involved in bacterial DNA replication, specifically DNA gyrase and topoisomerase IV [ 43 , 44 ]. However, E. coli can rapidly develop resistance to ciprofloxacin through mutations in genes linked to DNA replication [ 45 ]. In this case study, we used GeneScanner to highlight the most common amino acid substitutions associated with ciprofloxacin resistance in a natural population of E. coli isolates (n = 1509). Using GeneScanner, we analysed the DNA gyrase subunits A and B (GyrA, GyrB), and the topoisomerase IV subunits A and B (ParC and ParE) [ 45 ], across ciprofloxacin susceptible (n=1229) and resistant (n=280) isolates. To ensure accurate detection of group-specific variations, GeneScanner performs nucleotide and amino acid alignments before the populations are divided into susceptible and resistant groups, allowing the tool to accommodate insertions that may be unique to one group, and the resulting difference in alignment. Comparing the ProteinAnalysis sheets between groups, we identified several well-known amino acid changes at high frequencies in the resistant population. For example, over 97% of ciprofloxacin resistant isolates had GyrA substitutions at S83, and approximately 92% at D87 [ 45 ]. However, approximately 8% of the susceptible isolates also carried the S83 substitution, indicating that this change on its own may only confer small increases in resistance. Similarly, known resistance-associated ParC substitutions, such as S57, S80 and E84, were clearly overrepresented in our resistant population, suggesting that most isolates carried several resistance mutations. We further identified ParE I529L in 153 resistant isolates, a substitution associated previously associated with E. coli ST131 [ 46 ]. Interestingly, we also identified ParC substitution A192V in 151 of these isolates by comparing the ProteinMatrix, indicating co-variation between these otherwise well-conserved genes, and further highlighting the utility of GeneScanner. Case study 2: Vitamin B5 biosynthesis genes are conserved in Campylobacter jejuni isolated from cattle Campylobacter jejuni is a leading cause of bacterial gastroenteritis in humans, most often transmitted through the consumption of contaminated food, especially poultry [ 47 – 49 ]. C. jejuni colonises multiple hosts, including chickens, cattle, and wild birds, and distinct lineages frequently display host preference, consistent with niche specialisation through adaptation. One such adaptation involves the pantothenate (vitamin B5) biosynthesis pathway encoded by the panBCD locus. The first formal bacterial genome-wide association study (GWAS) [ 21 ] revealed that host association signals arose from two major forms of genetic variation. First, genes within the primary host-associated region were more commonly absent from chicken isolates. Second, when present, their sequences differed between cattle and chicken isolates, with cattle alleles exhibiting reduced homologous sequence variation – consistent with gene conservation. As a result, GWAS revealed host-association genes and alleles. We used GeneScanner in a second case study to validate these findings in a larger collection of C. jejuni ST-45 complex isolates, comprising 277 from cattle and an equal number from chickens. Consistent with previous results, the panBCD locus was more common in cattle isolates (69%) than in those from chickens (60%), as indicated on the output sheet. Additionally, chicken-associated isolates generally exhibited a greater diversity of unique alleles for the panBCD locus. This pattern was confirmed by GeneScanner, which confirmed the elevated overall genetic variation among chicken isolates. For instance, although the number of unique panB alleles was similar between cattle and chicken isolates, the 826 nucleotide panB alignment contained 18 variable sites in isolates from cattle, whereas chicken isolates had 80. A similar trend was observed for panC , with 21 vs. 88 variable sites, with only panD showing more comparable values with 26 vs 35 variable sites in cattle and chicken isolates respectively. These statistics are readily accessible in the Summary Statistics tab of the output files. Overall, GeneScanner analyses refined previous findings by pinpointing host-associated sites of genomic differentiation and conserved coding regions. This higher-resolution view confirmed the enrichment of the panBCD in cattle-associated lineages, while revealing markedly greater allelic diversity in chicken-associated lineages. Thus, GeneScanner strengthened the evidence for host-driven genomic signatures shaped by the interplay of metabolic capability, gene flow, and selective pressures in C. jejuni populations. Case study 3: Intergenic variation suggests strain-level modulation of biofilm expression The third GeneScanner case study focused on cross-species comparison of a non-coding but significant regulatory region in staphylococci. The i nter c ellular a dhesion ( ica) locus has an important role in biofilm formation in Staphylococcus aureus [ 50 , 51 ], contributing to persistent infections and treatment failure. The locus comprises icaADBC genes, which encode enzymes involved in polysaccharide intercellular adhesin (PIA) production, and the divergently transcribed repressor icaR. icaADBC expression is regulated by multiple elements beyond canonical transcription and translation initiation sites. Hence, the icaR _ icaA intergenic region contains binding sites for several regulatory proteins, including the repressors IcaR [ 52 ], TcaR [ 53 , 54 ], and Rob [ 36 ], and the activator SarA [ 56 ]. The presence of the 163 nucleotide-long intergenic region is highly conserved across a diverse S. aureus isolate population [ 57 ], but despite nucleotide-resolution analysis of regulator binding sites in model strains, the genetic variation of the icaR_icaA intergenic region remains understudied. GeneScanner nucleotide analysis revealed multiple loci with high levels of genetic variability within the ica promoter region, particularly near the start codons of icaR and icaA , as well as within the IcaR binding site ( Fig. 6 ). Notably, we also observed variation within the Rob binding site, including modifications and complete loss of the TATTT motif ( Fig. 6 , orange box), which is essential for Rob recognition and binding. Such changes are likely to impair Rob-mediated repression, potentially leading to increased ica expression and enhanced biofilm formation. These patterns of variation suggest the potential for strain-specific modulation of ica operon activity and biofilm production. In contrast, several SarA binding sites appeared to be highly conserved across all S. aureus isolates analysed, indicating that SarA may serve as a consistent activator of ica expression across diverse genetic backgrounds. Consistent with previous reports, the ica operon was absent in more distantly related staphylococcal species [ 50 ]. In contrast, 482 out of 1000 S. epidermidis isolates harboured the ica operon. However, GeneScanner summary statistics revealed only five variable sites within the icaR_icaA intergenic region, compared to 58 in S. aureus . This indicates a markedly higher degree of regulatory intergenic variation in S. aureus . Together, these findings demonstrate the utility of GeneScanner for interspecies comparison of non-coding regions, providing insights into the evolution of regulatory elements and their potential impact on phenotypic diversity. Conclusion Extensive bacterial genome datasets provide major opportunities to study microbial diversity, evolution, and adaptation [ 58 ]. However, connecting mutations discovered in controlled laboratory experiments with the naturally occurring variation present in bacterial populations remains challenging [ 15 , 35 ]. This methodological disconnect is particularly pronounced for researchers without extensive bioinformatics expertise, as most available tools require complex computational skills and resources. GeneScanner is designed to bridge this gap by offering an accessible and powerful platform for analysing gene- and protein-level variation across large bacterial genome collections. The tool efficiently identifies both nucleotide variation and protein-altering amino acid changes in genes and proteins associated with phenotypic traits, including antibiotic resistance studies and pathogen surveillance. Demonstrating the functionality and broad applicability of GeneScanner, we applied it to designed and simulated datasets as well as three distinct bacterial datasets encompassing different species, phenotypes, and gene loci. In each case, the tool successfully identified nucleotide-level and protein-altering changes associated with relevant traits, illustrating GeneScanner’s utility as a comprehensive and versatile platform for investigating genetic variation across a wide range of bacterial systems, traits, and experimental contexts. As with most genome analysis tools, the reliability of results depends heavily on data quality and sampling design. While GeneScanner includes quality control, poor assemblies, incomplete genomes, or incorrect annotations may introduce artefacts. Furthermore, biased sampling can distort allele frequencies, for example, overrepresentation of certain lineages can generate genotype-phenotype associations driven by population structure rather than true causality. To address this, analyses can incorporate population subsampling [ 28 , 59 ], linear mixed models [ 60 , 61 ], and phylogenetic approaches [ 62 ] to account for the clonal structure of the population. Additionally, using high-quality, well-annotated genomes and including metadata on isolation source, geographic origin, and collection date can help control for potential confounding factors. Uncovering the structural and functional impacts of naturally occurring genetic variation will be essential for advancing our understanding of microbial adaptation and phenotype variation. GeneScanner provides a powerful starting point by efficiently identifying candidate SNPs and amino acid changes across large bacterial genome collections. Of course, understanding how these alterations influence protein conformation, stability, or activity will require additional analysis. Integrating structural bioinformatics methods - such as domain conservation analysis [ 63 , 64 ], structure–function prediction [ 65 – 67 ], and molecular dynamics [ 68 , 69 ] - alongside experimental validation could achieve this. Advancing microbiology in this direction will move future research beyond merely detecting variation, toward mechanistically explaining how it drives microbial phenotypes, maximizing the impact of bacterial genomic data on surveillance, outbreak response, and therapeutic development. Methods Technical specifications All analyses were executed on PubMLST using the GeneScanner wrapper plugin [ 20 ]. Core software versions were Python 3.6+, Biopython 1.84 [ 70 ], pandas 2.2.2 [ 71 ], XlsxWriter 3.2.0 [ 72 ], MAFFT 7.490 [ 73 ], and snp-sites 2.5.1 [ 37 ]. Detailed requirement for each release version and all code are available in the public GitHub repository ( https://github.com/Sheppard-Lab/GeneScanner ). GeneScanner pipeline and analytical framework Multiple-sequence alignment FASTA files are the primary input for GeneScanner. Two quality filters are applied relative to a designated reference sequence: a default minimum of 80% ungapped coverage across the alignment length and at least 80% pairwise identity. If requested, a VCF file is generated via snp-sites [ 37 ]. In nucleotide analysis mode, GeneScanner iterates through the alignment position-by-position while preserving codon phase relative to the reference, which can be user-defined or automatically set as the first sequence in the alignment. For each non-gap codon triplet, the translated codon is compared with the reference translation to classify synonymous and non-synonymous mutations. Insertions, deletions, and premature stop codons are recorded independently. When protein analysis mode is activated, GeneScanner first removes all the gaps within the original alignment file, translates in the user-specified frame and realigns peptides with MAFFT [ 73 ]. To prevent alignment disruption caused by early stop codons, the program temporarily fills post-stop-codon regions with reference residues, which are removed after realignment to maintain alignment quality. Optional grouping analyses are triggered when the user provides a two-column CSV file specifying isolate names and their categories. GeneScanner partitions the dataset by category and re-runs the full analytical pipeline for each subset, producing parallel mutation analysis reports and mutation matrix worksheets that share column definitions but are restricted to their respective groups. When distinct reference sequences are supplied for individual groups, GeneScanner automatically reassigns the appropriate reference prior to variant calling. After all analyses are complete, GeneScanner compiles the results into a spreadsheet (.xlsx) by using pandas [ 71 ] and XlsxWriter [ 72 ]. All parameters are user-configurable via command-line flags and run metadata as well as quality-control filtering logs can be written to separate files. Synthetic sequence design for GeneScanner benchmarking To create a dataset with explicitly designed mutations, we generated a 300 bp reference coding sequence and designed point mutations and short indels using custom Python code from the GeneScanner repository. To evaluate coding-effect categorization, we placed synonymous substitutions at alignment positions 12 and 147 with allele frequencies of 0.20 and 0.30, respectively; nonsynonymous substitutions at positions 46 and 181 with frequencies of 0.20 and 0.30, and a nonsense stop codon substitution at position 210 with frequency 0.10 to confirm that GeneScanner treats stop-gains distinctly from other variants. To assess frameshift handling, we designed two indel scenarios with single-nucleotide changes: (i) frameshift leading to premature termination via a deletion at nucleotide position 4 (10%) and an insertion at position 69 (10%) and (ii) a frame-restoring pair consisting of an insertion at position 153 (5%) followed by a downstream deletion at position 219 (5%) such that the reading frame is restored after a segment of shifted translation. All coordinates are based on the aligned nucleotide sequence and amino-acid coordinates refer to the translation of the reference. In silico sequence evolution simulation for GeneScanner validation Sequence evolution of two subpopulations was simulated under opposing selection regimes using SLiM v5.0 [ 74 ] with nucleotide-based models and phylogeny data. The model script was generated applying the previous published code [ 75 ]. The simulated genome comprised a 300 bp random ancestral coding sequence. Mutations followed a symmetric Jukes-Cantor model with a per-site mutation rate of μ = 1 × 10 −6 per generation. Recombination occurred at a rate of ρ = 1/3 × 10 −4 per site per generation along the locus with 30bp as average recombination block length. At generation 1, we split the population into two panmictic subpopulations of 5000 haploid individuals each. Selection was restricted to a 31 bp window spanning nucleotide positions 100 - 130. Fitness was calculated by multiplying over sites and across the two populations. We assigned opposite selection coefficients in the two subpopulations: In subpopulation p1, mutations within the window were beneficial (s = + 0.005), whereas in subpopulation p2, the same mutations were deleterious (s = −0.005). Mutations outside the window mimicked neutral mutation with slightly deleterious fitness cost (s = −0.0001). Total fitness is calculated by 1 +∑ s compared to 1 at starting reference gene. To limit the extent of linkage hitchhiking, we chose an elevated recombination rate, as theory predicts that the influence of a selective sweep on linked diversity decreases with recombination distance relative to selection strength [ 76 ]. Density dependent fitness adjustment was added to keep the population size near the original 5,000 haploid genomes per population while it changes the sequence frequency within populations. Simulations were run for 1,000 generations under SLiM’s default non-Wright–Fisher framework. At the final generation, 500 sequences were sampled from each subpopulation. Sampled sequences were aligned using MAFFT, and mutational patterns were subsequently characterized using GeneScanner in dual nucleotide– protein mode. Case study 1: genetic variation underlying ciprofloxacin resistance in E. coli The E. coli genome assemblies (n=1509, Supplementary file 7) from a previous study [ 77 ] were accessed on the PubMLST database [ 20 ]. Information on ciprofloxacin susceptibility of these isolates was taken from supplementary material. Nucleotide sequences of query genes ( gyrA, gyrB, parC and parE ) were downloaded from GenBank and used to query against isolate assemblies using blastn (BLAST 2.12.0+ with settings: word size: 20; reward: 2; penalty: −3; gapopen: 5; gapextend: 2). The identified matching sequences were extracted and aligned using MAFFT v7.505 with default parameters to create the nucleotide alignment input file. GeneScanner was run in ‘nucleotide + protein’ mode, selecting strain E. coli K12 MG1655 (PubMLST Escherichia ID: 1167) as reference for both groups. A CSV file for group 2 (resistant isolates) was created. Selected data were, as mutation frequency and normalised by the isolate number, visualised using GraphPad Prism (version 10.5.0) Case study 2: Conservation and diversity of vitamin B5 biosynthesis genes in Campylobacter The genome assemblies of Campylobacter jejuni ST-45 complex isolates were accessed on the PubMLST database [ 20 ], using all available isolates resulting from the search term “cattle” (n=277), and randomly selecting an equal number of isolates containing the term “chicken” in the search field (Supplementary file 7). Nucleotide sequences of the panBCD locus intergenic were downloaded from GenBank and used to query against isolate assemblies using blastn (BLAST 2.12.0+ with settings: word size: 20; reward: 2; penalty: −3; gapopen: 5; gapextend: 2). The identified matching sequences were extracted and aligned using MAFFT v7.505 with default parameters to create the nucleotide alignment input file. GeneScanner was run in ‘nucleotide + protein’ mode, selecting strain Campylobacter jejuni NCTC11168 (PubMLST Campylobacter jejuni/coli ID: 48) as reference. A CSV file for group 2 (isolates from cattle) was created. The number of unique alleles was determined using the BIGSdb plugin GenomeComparator [ 78 ] using the same isolates as input and default settings. Selected data were, as mutation frequency and normalised by the isolate number, visualised using GraphPad Prism (version 10.5.0). Icons in Figure 5 were designed by PLANBstudio and sourced from flaticon.com. Download figure Open in new tab Figure 4. Ciprofloxacin resistance is linked to genetic variation in natural E. coli isolates. Genetic variation of ciprofloxacin resistance-related proteins was analysed in a natural population of 1509 E. coli isolates. Amino acid changes at each protein alignment position for GyrA (A+B), GyrB (C+D), ParC (E+F), and ParE (G+H), are given in blue or red, to indicate the ciprofloxacin susceptible (n = 1229), and ciprofloxacin resistant (n = 280) subpopulations, respectively. Download figure Open in new tab Figure 5. Conservation and diversity of vitamin B5 biosynthesis genes in Campylobacter jejuni from cattle and chickens. A: The panBCD locus is present in a higher proportion of cattle isolates compared to chicken isolates. B: Chicken-associated isolates harbour a greater number of unique panBCD alleles relative to cattle isolates. C-G: Patterns of genetic variation across the panBCD locus show reduced diversity in cattle isolates compared with chicken isolates. Download figure Open in new tab Figure 6. Genetic variation at the non-coding regulatory regions of the biofilm-associated ica operon in S. aureus and S. epidermidis . Single nucleotide polymorphisms (grey), insertions (green) and deletions (orange) are indicated along the nucleotide alignment (grey bar below graph) of the non-coding region between icaR and icaA in S. aureus (A) and S. epidermidis (B). Graphical representation of regulatory elements to scale: black: Shine-Dalgarno sequence for icaR and icaA , respectively; dark grey: likely −35 and −10 regions for transcription initiation of icaADBC operon; red: IcaR binding site; orange: Rob binding site with TATTT motif; yellow: TcaR binding sites; blue: SarA binding sites. Case study 3: Genetic variation in biofilm-regulating intergenic region in staphylococci The genome assemblies of S. aureus (n=1984), S. epidermidis (n=1000), S. hominis (n=13), S. haemolyticus (n=44), S. pseudintermedius (n=178) and S. chromogenes (n=49) were accessed on the PubMLST database [ 20 ] (Supplementary file 7). Nucleotide sequences of the icaR _ icaA intergenic region of S. aureus and S. epidermidis were downloaded from GenBank and used to query against isolate assemblies using using blastn (BLAST 2.12.0+ with settings: word size: 20; reward: 2; penalty: −3; gapopen: 5; gapextend: 2). The identified matching sequences were extracted and aligned using MAFFT v7.505 with default parameters to create the nucleotide alignment input file. GeneScanner was run in ‘nucleotide’ mode, selecting strain USA300_FPR3757 (PubMLST Staphylococcus aureus ID: 37463) and N13018T (PubMLST Staphylococcus epidermidis ID: 41156) as reference for S. aureus and S. epidermidis , respectively. As the NucleotideAnalysis sheet will still assume a coding sequence, the descriptions synonymous, non-synonymous and stop codons , should be ignored and mutation frequencies added together. Mutation frequencies were normalised to the isolate number and visualised using GraphPad Prism (version 10.5.0). Information on the regulator binding sites of icaR _ icaA intergenic region was taken from previous studies [ 50 , 52 – 55 ]. Data availability GeneScanner code and detailed requirements for each release version are publicly available ( https://github.com/Sheppard-Lab/GeneScanner , DOI: 10.5281/zenodo.17495646. GeneScanner requires Python v3.6+. GeneScanner is also available as plugin in the PubMLST online database ( https://bigsdb.readthedocs.io/en/latest/data_analysis/genescanner.html ). All genome assemblies used are available through the PubMLST database. Isolate IDs, query sequences, and alignment and output files for designed and synthetic datasets can be found in the supplementary files. Author contributions Conceptualization: SKS, CMK Design & Methodology: SK, PSR, CMK, SKS, BAS Implementation: SK, PSR, BAS, KJ Analysis: CMK, SK, PSR, SKS Data Curation: CMK, SK, PSR, BAS, KJ Writing - Original Draft: CMK, SK, PSR, SKS Writing - Review & Editing: all authors Visualization: CMK, SK Supervision: SKS, CMK Project administration: SKS, CMK Funding acquisition: SKS; CMK Funding CMK was funded by an EPA Junior Research Fellowship from Linacre College Oxford, a University of Oxford Medical Sciences Internal Fund Pump-priming Award (0015060), and a BBSRC Fellowship (UKRI905). KAJ was funded by a Wellcome Trust Biomedical Resource Grant (grant number 218205/Z/19/Z). PSR is funded through an Ineos Oxford Institute (IOI) DPhil Studentship. SKS was supported by an IOI grant; Wellcome Trust grant 088786/C/09/Z, and UKRI grants MR/L015080/1, MR/V001213/1, MR/S009264/1, and MR/T030062/1. Competing interests The authors declare that they have no competing interests. Funder Information Declared UK Research and Innovation, https://ror.org/001aqnf71 , UKRI905 , MR/L015080/1 , MR/V001213/1 , MR/S009264/1 , MR/T030062/1 Wellcome Trust , 218205/Z/19/Z , 088786/C/09/Z Footnotes https://github.com/Sheppard-Lab/GeneScanner References [1]. ↵ Baba T , Ara T , Hasegawa M , Takai Y , Okumura Y , Baba M , et al. Construction of Escherichia coli K-12 in -frame, singlegene knockout mutants: the Keio collection . Molecular Systems Biology 2006 . doi: 10.1038/msb4100050 . OpenUrl Abstract / FREE Full Text [2]. Berardinis V de , Vallenet D , Castelli V , Besnard M , Pinet A , Cruaud C , et al. A complete collection of singlegene deletion mutants of Acinetobacter baylyi ADP1 . Molecular Systems Biology 2008 . doi: 10.1038/msb.2008.10 . OpenUrl Abstract / FREE Full Text [3]. Porwollik S , Santiviago CA , Cheng P , Long F , Desai P , Fredlund J , et al. Defined Single-Gene and Multi-Gene Deletion Mutant Collections in Salmonella enterica sv Typhimurium . PLOS ONE 2014 ; 9 : e99820 . doi: 10.1371/journal.pone.0099820 . OpenUrl CrossRef PubMed [4]. ↵ Koo B-M , Kritikos G , Farelli JD , Horia Todor , Tong K , Kimsey H , et al. Construction and Analysis of Two Genome-Scale Deletion Libraries for Bacillus subtilis . Cels 2017 ; 4 : 291 – 305 .e7. doi: 10.1016/j.cels.2016.12.013 . OpenUrl CrossRef PubMed [5]. ↵ van Opijnen T , Bodi KL , Camilli A. Tn-seq: high-throughput parallel sequencing for fitness and genetic interaction studies in microorganisms . Nat Methods 2009 ; 6 : 767 – 72 . doi: 10.1038/nmeth.1377 . OpenUrl CrossRef PubMed Web of Science [6]. Langridge GC , Phan M-D , Turner DJ , Perkins TT , Parts L , Haase J , et al. Simultaneous assay of every Salmonella Typhi gene using one million transposon mutants . Genome Res 2009 ; 19 : 2308 – 16 . doi: 10.1101/gr.097097.109 . OpenUrl Abstract / FREE Full Text [7]. Gawronski JD , Wong SMS , Giannoukos G , Ward DV , Akerley BJ . Tracking insertion mutants within libraries by deep sequencing and a genome-wide screen for Haemophilus genes required in the lung . Proceedings of the National Academy of Sciences 2009 ; 106 : 16422 – 7 . doi: 10.1073/pnas.0906627106 . OpenUrl Abstract / FREE Full Text [8]. Goodman AL , McNulty NP , Zhao Y , Leip D , Mitra RD , Lozupone CA , et al. Identifying Genetic Determinants Needed to Establish a Human Gut Symbiont in Its Habitat . Cell Host & Microbe 2009 ; 6 : 279 – 89 . doi: 10.1016/j.chom.2009.08.003 . OpenUrl CrossRef PubMed Web of Science [9]. van Opijnen T , Camilli A. Transposon insertion sequencing: a new tool for systems-level analysis of microorganisms . Nat Rev Microbiol 2013 ; 11 : 435 – 42 . doi: 10.1038/nrmicro3033 . OpenUrl CrossRef PubMed [10]. ↵ Cain AK , Barquist L , Goodman AL , Paulsen IT , Parkhill J , van Opijnen T. A decade of advances in transposon-insertion sequencing . Nat Rev Genet 2020 ; 21 : 526 – 40 . doi: 10.1038/s41576-020-0244-x . OpenUrl CrossRef PubMed [11]. ↵ Conrad TM , Lewis NE , Palsson BØ. Microbial laboratory evolution in the era of genome - scale science . Molecular Systems Biology 2011 . doi: 10.1038/msb.2011.42 . OpenUrl Abstract / FREE Full Text [12]. Baym M , Lieberman TD , Kelsic ED , Chait R , Gross R , Yelin I , et al. Spatiotemporal microbial evolution on antibiotic landscapes . Science 2016 . doi: 10.1126/science.aag0822 . OpenUrl Abstract / FREE Full Text [13]. Lenski RE . Experimental evolution and the dynamics of adaptation and genome evolution in microbial populations . ISME J 2017 ; 11 : 2181 – 94 . doi: 10.1038/ismej.2017.69 . OpenUrl CrossRef PubMed [14]. ↵ Maeda T , Iwasawa J , Kotani H , Sakata N , Kawada M , Horinouchi T , et al. Highthroughput laboratory evolution reveals evolutionary constraints in Escherichia coli . Nat Commun 2020 ; 11 : 1 – 13 . doi: 10.1038/s41467-020-19713-w . OpenUrl CrossRef PubMed [15]. ↵ Ascensao JA , Desai MM . Experimental evolution in an era of molecular manipulation . Nat Rev Genet 2025 : 1 – 15 . doi: 10.1038/s41576-025-00867-6 . OpenUrl CrossRef [16]. ↵ O’Cathail C , Ahamed A , Burgin J , Cummins C , Devaraj R , Gueye K , et al. The European Nucleotide Archive in 2024 . Nucleic Acids Res 2025 ; 53 : D49 – 55 . doi: 10.1093/nar/gkae975 . OpenUrl CrossRef PubMed [17]. ↵ Sayers EW , Beck J , Bolton EE , Brister JR , Chan J , Connor R , et al. Database resources of the National Center for Biotechnology Information in 2025 . Nucleic Acids Res 2025 ; 53 : D20 – 9 . doi: 10.1093/nar/gkae979 . OpenUrl CrossRef [18]. ↵ Zhou Z , Alikhan N-F , Mohamed K , Fan Y , Group the AS , Achtman M. The EnteroBase user’s guide, with case studies on Salmonella transmissions, Yersinia pestis phylogeny, and Escherichia core genomic diversity . Genome Research 2020 ; 30 : 138 . doi: 10.1101/gr.251678.119 . OpenUrl Abstract / FREE Full Text [19]. Dyer NP , Päuker B , Baxter L , Gupta A , Bunk B , Overmann J , et al. EnteroBase in 2025: exploring the genomic epidemiology of bacterial pathogens . Nucleic Acids Res 2025 ; 53 : D757 – 62 . doi: 10.1093/nar/gkae902 . OpenUrl CrossRef PubMed [20]. ↵ Jolley KA , Bray JE , Maiden MCJ . Open-access bacterial population genomics: BIGSdb software, the PubMLST.org website and their applications . Wellcome Open Research 2018 ; 3 . doi: 10.12688/wellcomeopenres.14826.1 . OpenUrl CrossRef PubMed [21]. ↵ Sheppard SK , Didelot X , Meric G , Torralbo A , Jolley KA , Kelly DJ , et al. Genome-wide association study identifies vitamin B 5 biosynthesis as a host specificity factor in Campylobacter . Proceedings of the National Academy of Sciences 2013 ; 110 : 11923 – 7 . doi: 10.1073/pnas.1305559110 . OpenUrl Abstract / FREE Full Text [22]. Yahara K , Méric G , Taylor AJ , Vries SPW de , Murray S , Pascoe B , et al. Genome-wide association of functional traits linked with Campylobacter jejuni survival from farm to fork . Environmental Mirobiology 2016 . [23]. ↵ Taylor AJ , Yahara K , Pascoe B , Ko S , Mageiros L , Mourkas E , et al. Epistasis, coregenome disharmony, and adaptation in recombining bacteria . mBio 2024 . doi: 10.1128/mbio.00581-24 . OpenUrl CrossRef PubMed [24]. ↵ Hwang W , Yong JH , Min KB , Lee K-M , Pascoe B , Sheppard SK , et al. Genome-wide association study of signature genetic alterations among pseudomonas aeruginosa cystic fibrosis isolates . PLOS Pathogens 2021 ; 17 : e1009681 . doi: 10.1371/journal.ppat.1009681 . OpenUrl CrossRef PubMed [25]. ↵ Laabei M , Recker M , Rudkin JK , Aldeljawi M , Gulay Z , Sloan TJ , et al. Predicting the virulence of MRSA from its genome sequence . Genome Res 2014 ; 24 : 839 – 49 . doi: 10.1101/gr.165415.113 . OpenUrl Abstract / FREE Full Text [26]. Laabei M , Uhlemann A-C , Lowy FD , Austin ED , Yokoyama M , Ouadi K , et al. Evolutionary trade-offs underlie the multi-faceted virulence of Staphylococcus aureus . PLOS Biology 2015 ; 13 : e1002229 . doi: 10.1371/journal.pbio.1002229 . OpenUrl CrossRef PubMed [27]. Berthenet E , Yahara K , Thorell K , Pascoe B , Meric G , Mikhail JM , et al. A GWAS on Helicobacter pylori strains points to genetic variants associated with gastric cancer risk . BMC Biol 2018 ; 16 : 1 – 11 . doi: 10.1186/s12915-018-0550-3 . OpenUrl CrossRef PubMed [28]. ↵ Méric G , Mageiros L , Pensar J , Laabei M , Yahara K , Pascoe B , et al. Disease-associated genotypes of the commensal skin bacterium Staphylococcus epidermidis . Nat Commun 2018 ; 9 : 1 – 11 . doi: 10.1038/s41467-018-07368-7 . OpenUrl CrossRef PubMed [29]. Monteith W , Pascoe B , Mourkas E , Clark J , Hakim M , Hitchings MD , et al. Contrasting genes conferring short- and long-term biofilm adaptation in Listeria . Microbial Genomics 2023 ; 9 : 001114 . doi: 10.1099/mgen.0.001114 . OpenUrl CrossRef PubMed [30]. ↵ Pascoe B , Méric G , Murray S , Yahara K , Mageiros L , Bowen R , et al. Enhanced biofilm formation and multihost transmission evolve from divergent genetic backgrounds in Campylobacter jejuni . Environmental Microbiology 2015 . [31]. ↵ Farhat MR , Shapiro BJ , Kieser KJ , Sultana R , Jacobson KR , Victor TC , et al. Genomic analysis identifies targets of convergent positive selection in drug-resistant Mycobacterium tuberculosis . Nat Genet 2013 ; 45 : 1183 – 9 . doi: 10.1038/ng.2747 . OpenUrl CrossRef PubMed [32]. Chewapreecha C , Marttinen P , Croucher NJ , Salter SJ , Harris SR , Mather AE , et al. Comprehensive Identification of Single Nucleotide Polymorphisms Associated with Betalactam Resistance within Pneumococcal Mosaic Genes . PLOS Genetics 2014 ; 10 : e1004547 . doi: 10.1371/journal.pgen.1004547 . OpenUrl CrossRef PubMed [33]. Farhat MR , Freschi L , Calderon R , Ioerger T , Snyder M , Meehan CJ , et al. GWAS for quantitative resistance phenotypes in Mycobacterium tuberculosis reveals resistance genes and regulatory regions . Nat Commun 2019 ; 10 : 1 – 11 . doi: 10.1038/s41467-019-10110-6 . OpenUrl CrossRef PubMed [34]. ↵ Ma KC , Mortimer TD , Duckett MA , Hicks AL , Wheeler NE , Sánchez-Busó L , et al. Increased power from conditional bacterial genome-wide association identifies macrolide resistance mutations in Neisseria gonorrhoeae . Nat Commun 2020 ; 11 : 1 – 8 . doi: 10.1038/s41467-020-19250-6 . OpenUrl CrossRef PubMed [35]. ↵ Kobras CM , Fenton AK , Sheppard SK . Next-generation microbiology: from comparative genomics to gene function . Genome Biol 2021 ; 22 : 1 – 16 . doi: 10.1186/s13059-021-02344-9 . OpenUrl CrossRef PubMed [36]. ↵ Kobras CM , Monteith W , Somerville S , Delaney JM , Khan I , Brimble C , et al. Loss of Pde1 function acts as an evolutionary gateway to penicillin resistance in Streptococcus pneumoniae . Proceedings of the National Academy of Sciences 2023 ; 120 : e2308029120 . doi: 10.1073/pnas.2308029120 . OpenUrl CrossRef PubMed [37]. ↵ Page AJ , Taylor B , Delaney AJ , Soares J , Seemann T , Keane JA , et al. SNP-sites: rapid efficient extraction of SNPs from multi-FASTA alignments . Microbial Genomics 2016 ; 2 : e000056 . doi: 10.1099/mgen.0.000056 . OpenUrl CrossRef PubMed [38]. Lindenbaum P. JVarkit: java-based utilities for Bioinformatics 2015 :347403 Bytes. doi: 10.6084/M9.FIGSHARE.1425030 . OpenUrl CrossRef [39]. Capella-Gutiérrez S , Silla-Martínez JM , Gabaldón T. trimAl: a tool for automated alignment trimming in large-scale phylogenetic analyses . Bioinformatics 2009 ; 25 : 1972 – 3 . doi: 10.1093/bioinformatics/btp348 . OpenUrl CrossRef PubMed Web of Science [40]. Lischer HEL , Excoffier L. PGDSpider: an automated data conversion tool for connecting population genetics and genomics programs . Bioinformatics 2012 ; 28 : 298 – 9 . doi: 10.1093/bioinformatics/btr642 . OpenUrl CrossRef PubMed Web of Science [41]. ↵ Swofford DL . Paup* . Phylogenetic Analysis Using Parsimony (* and Other Methods). (No Title) 2002 . [42]. ↵ Murray CJL , Ikuta KS , Sharara F , Swetschinski L , Aguilar GR , Gray A , et al. Global burden of bacterial antimicrobial resistance in 2019: a systematic analysis . The Lancet 2022 ; 399 : 629 – 55 . doi: 10.1016/S0140-6736(21)02724-0 . OpenUrl CrossRef PubMed [43]. ↵ Khodursky AB , Zechiedrich EL , Cozzarelli NR . Topoisomerase IV is a target of quinolones in Escherichia coli . Proceedings of the National Academy of Sciences 1995 ; 92 : 11801 – 5 . doi: 10.1073/pnas.92.25.11801 . OpenUrl Abstract / FREE Full Text [44]. ↵ Drlica K. Mechanism of fluoroquinolone action . Current Opinion in Microbiology 1999 ; 2 : 504 – 8 . doi: 10.1016/S1369-5274(99)00008-9 . OpenUrl CrossRef PubMed Web of Science [45]. ↵ Hopkins KL , Davies RH , Threlfall EJ . Mechanisms of quinolone resistance in Escherichia coli and Salmonella: Recent developments . International Journal of Antimicrobial Agents 2005 ; 25 : 358 – 73 . doi: 10.1016/j.ijantimicag.2005.02.006 . OpenUrl CrossRef PubMed Web of Science [46]. ↵ Paltansing S , Kraakman MEM , Ras JMC , Wessels E , Bernards AT . Characterization of fluoroquinolone and cephalosporin resistance mechanisms in Enterobacteriaceae isolated in a Dutch teaching hospital reveals the presence of an Escherichia coli ST131 clone with a specific mutation in parE . J Antimicrob Chemother 2013 ; 68 : 40 – 5 . doi: 10.1093/jac/dks365 . OpenUrl CrossRef PubMed Web of Science [47]. ↵ Sheppard SK , Dallas JF , Strachan NJC , MacRae M , McCarthy ND , Wilson DJ , et al. Campylobacter genotyping to determine the source of human infection . Clinical Infectious Diseases 2009 ; 48 : 1072 – 8 . OpenUrl CrossRef PubMed Web of Science [48]. Arning N , Sheppard SK , Bayliss S , Clifton DA , Wilson DJ . Machine learning to predict the source of campylobacteriosis using whole genome data . PLOS Genetics 2021 ; 17 : e1009436 . doi: 10.1371/journal.pgen.1009436 . OpenUrl CrossRef PubMed [49]. ↵ Pascoe B , Futcher G , Pensar J , Bayliss SC , Mourkas E , Calland JK , et al. Machine learning to attribute the source of Campylobacter infections in the United States: A retrospective analysis of national surveillance data . Journal of Infection 2024 ; 89 . doi: 10.1016/j.jinf.2024.106265 . OpenUrl CrossRef PubMed [50]. ↵ Cramton SE , Gerke C , Schnell NF , Nichols WW , Götz F. The intercellular adhesion (ica) locus is present in Staphylococcus aureus and is required for biofilm formation . Infection and Immunity 1999 ; 67 : 5427 . doi: 10.1128/iai.67.10.5427-5433.1999 . OpenUrl Abstract / FREE Full Text [51]. ↵ O’Gara JP . ica and beyond: biofilm mechanisms and regulation in Staphylococcus epidermidis and Staphylococcus aureus . FEMS Microbiol Lett 2007 ; 270 : 179 – 88 . doi: 10.1111/j.1574-6968.2007.00688.x . OpenUrl CrossRef PubMed [52]. ↵ Conlon KM , Humphreys H , O’Gara JP . icaR encodes a transcriptional repressor involved in environmental regulation of icaR operon expression and biofilm formation in Staphylococcus epidermidis . Journal of Bacteriology 2002 . doi: 10.1128/jb.184.16.4400-4408.2002 . OpenUrl CrossRef [53]. ↵ Jefferson KK , Pier DB , Goldmann DA , Pier GB . The teicoplanin-associated locus regulator (TcaR) and the intercellular adhesin locus regulator (IcaR) are transcriptional inhibitors of the ica locus in Staphylococcus aureus . Journal of Bacteriology 2004 . doi: 10.1128/jb.186.8.2449-2456.2004 . OpenUrl CrossRef [54]. ↵ Chang Y-M , Jeng W-Y , Ko T-P , Yeh Y-J , Chen CK-M , Wang AH-J. Structural study of TcaR and its complexes with multiple antibiotics from Staphylococcus epidermidis . Proceedings of the National Academy of Sciences 2010 ; 107 : 8617 – 22 . doi: 10.1073/pnas.0913302107 . OpenUrl Abstract / FREE Full Text [55]. ↵ Yu L , Hisatsune J , Hayashi I , Tatsukawa N , Sato’o Y , Mizumachi E , et al. A Novel Repressor of the ica Locus Discovered in Clinically Isolated Super-Biofilm-Elaborating Staphylococcus aureus . mBio 2017 . doi: 10.1128/mbio.02282-16 . OpenUrl CrossRef [56]. ↵ Tormo MÁ , Martí M , Valle J , Manna AC , Cheung AL , Lasa I , et al. SarA is an essential positive regulator of Staphylococcus epidermidis biofilm development . Journal of Bacteriology 2005 . doi: 10.1128/jb.187.7.2348-2356.2005 . OpenUrl CrossRef [57]. ↵ Young BC , Wu C-H , Charlesworth J , Earle S , Price JR , Gordon NC , et al. Antimicrobial resistance determinants are associated with Staphylococcus aureus bacteraemia and adaptation to the healthcare environment: a bacterial genome-wide association study . Microbial Genomics 2021 ; 7 : 000700 . doi: 10.1099/mgen.0.000700 . OpenUrl CrossRef PubMed [58]. ↵ Hall N. Advanced sequencing technologies and their wider impact in microbiology . J Exp Biol 2007 ; 210 : 1518 – 25 . doi: 10.1242/jeb.001370 . OpenUrl Abstract / FREE Full Text [59]. ↵ Mageiros L , Méric G , Bayliss SC , Pensar J , Pascoe B , Mourkas E , et al. Genome evolution and the emergence of pathogenicity in avian Escherichia coli . Nature Communications 2021 ; 12 : 765 . doi: 10.1038/s41467-021-20988-w . OpenUrl CrossRef PubMed [60]. ↵ Lees JA , Vehkala M , Välimäki N , Harris SR , Chewapreecha C , Croucher NJ , et al. Sequence element enrichment analysis to determine the genetic basis of bacterial phenotypes . Nat Commun 2016 ; 7 : 1 – 8 . doi: 10.1038/ncomms12797 . OpenUrl CrossRef PubMed [61]. ↵ Earle SG , Wu C-H , Charlesworth J , Stoesser N , Gordon NC , Walker TM , et al. Identifying lineage effects when controlling for population structure improves power in bacterial association studies . Nat Microbiol 2016 ; 1 : 1 – 8 . doi: 10.1038/nmicrobiol.2016.41 . OpenUrl CrossRef [62]. ↵ Collins C , Didelot X. A phylogenetic method to perform genome-wide association studies in microbes that accounts for population structure and recombination . PLOS Computational Biology 2018 ; 14 : e1005958 . doi: 10.1371/journal.pcbi.1005958 . OpenUrl CrossRef PubMed [63]. ↵ Marchler-Bauer A , Anderson JB , Derbyshire MK , DeWeese-Scott C , Gonzales NR , Gwadz M , et al. CDD: a conserved domain database for interactive domain family analysis . Nucleic Acids Res 2007 ; 35 : D237 – 40 . doi: 10.1093/nar/gkl951 . OpenUrl CrossRef PubMed Web of Science [64]. ↵ Yang M , Derbyshire MK , Yamashita RA ,; archeler-Bauer A. NCBI’s Conserved Domain Database and Tools for Protein Domain Analysis - Yang - 2020 - Current Protocols in Bioinformatics - Wiley Online Library . Current Protocols in Bioinformatics 2019 . [65]. ↵ Yang J , Yan R , Roy A , Xu D , Poisson J , Zhang Y. The I-TASSER Suite: protein structure and function prediction . Nat Methods 2015 ; 12 : 7 – 8 . doi: 10.1038/nmeth.3213 . OpenUrl CrossRef PubMed [66]. Senior AW , Evans R , Jumper J , Kirkpatrick J , Sifre L , Green T , et al. Improved protein structure prediction using potentials from deep learning . Nature 2020 ; 577 : 706 – 10 . doi: 10.1038/s41586-019-1923-7 . OpenUrl CrossRef PubMed [67]. ↵ Jumper J , Evans R , Pritzel A , Green T , Figurnov M , Ronneberger O , et al. Highly accurate protein structure prediction with AlphaFold . Nature 2021 ; 596 : 583 – 9 . doi: 10.1038/s41586-021-03819-2 . OpenUrl CrossRef PubMed [68]. ↵ Cheatham TE , Kollman PA . Molecular Dynamics Simulation of Nucleic Acids . Annual Review of Physical Chemistry 2000 ; 51 : 435 – 71 . doi: 10.1146/annurev.physchem.51.1.435 . OpenUrl CrossRef PubMed Web of Science [69]. ↵ Khalid S , Brandner AF , Juraschko N , Newman KE , Pedebos C , Prakaash D , et al. Computational microbiology of bacteria: Advancements in molecular dynamics simulations . Structure 2023 ; 31 : 1320 – 7 . doi: 10.1016/j.str.2023.09.012 . OpenUrl CrossRef [70]. ↵ Cock PJA , Antao T , Chang JT , Chapman BA , Cox CJ , Dalke A , et al. Biopython: freely available Python tools for computational molecular biology and bioinformatics . Bioinformatics 2009 ; 25 : 1422 – 3 . doi: 10.1093/bioinformatics/btp163 . OpenUrl CrossRef PubMed Web of Science [71]. ↵ McKinney W. Data Structures for Statistical Computing in Python , Austin, Texas : 2010 , p. 56 – 61 . doi: 10.25080/Majora-92bf1922-00a . OpenUrl CrossRef [72]. ↵ xlsxwriter . PyPI 2025 . https://pypi.org/project/xlsxwriter/ (accessed June 18, 2025 ). [73]. ↵ Katoh K , Misawa K , Kuma K , Miyata T. MAFFT: a novel method for rapid multiple sequence alignment based on fast Fourier transform . Nucleic Acids Res 2002 ; 30 : 3059 – 66 . doi: 10.1093/nar/gkf436 . OpenUrl CrossRef PubMed Web of Science [74]. ↵ Messer PW . SLiM: Simulating Evolution with Selection and Linkage . Genetics 2013 ; 194 : 1037 . doi: 10.1534/genetics.113.152181 . OpenUrl Abstract / FREE Full Text [75]. ↵ Cury J , Haller BC , Achaz G , Jay F. Simulation of bacterial populations with SLiM . Peer Community Journal 2022 ; 2 . doi: 10.24072/pcjournal.72 . OpenUrl CrossRef [76]. ↵ Stephan W. Genetic hitchhiking versus background selection: the controversy and its implications . Philosophical Transactions of the Royal Society B: Biological Sciences 2010 . doi: 10.1098/rstb.2009.0278 . OpenUrl CrossRef PubMed [77]. ↵ Kallonen T , Brodrick HJ , Harris SR , Corander J , Brown NM , Martin V , et al. Systematic longitudinal survey of invasive Escherichia coli in England demonstrates a stable population structure only transiently disturbed by the emergence of ST131 . Genome Res 2017 ; 27 : 1437 – 49 . doi: 10.1101/gr.216606.116 . OpenUrl Abstract / FREE Full Text [78]. ↵ Jolley KA , Maiden MC . BIGSdb: Scalable analysis of bacterial genome variation at the population level . BMC Bioinformatics 2010 ; 11 : 1 – 11 . doi: 10.1186/1471-2105-11-595 . OpenUrl CrossRef PubMed View the discussion thread. Back to top Previous Next Posted November 04, 2025. Download PDF Supplementary Material Data/Code Email Thank you for your interest in spreading the word about bioRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following GeneScanner: profiling genetic variation across bacterial populations Message Subject (Your Name) has forwarded a page to you from bioRxiv Message Body (Your Name) thought you would like to see this page from the bioRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share GeneScanner: profiling genetic variation across bacterial populations Carolin M. Kobras , Seungwon Ko , Priyanshu S. Raikwar , Broncio Aguilar-Sanjuan , Keith A. Jolley , Samuel K. Sheppard bioRxiv 2025.11.02.685864; doi: https://doi.org/10.1101/2025.11.02.685864 Share This Article: Copy Citation Tools GeneScanner: profiling genetic variation across bacterial populations Carolin M. Kobras , Seungwon Ko , Priyanshu S. Raikwar , Broncio Aguilar-Sanjuan , Keith A. Jolley , Samuel K. Sheppard bioRxiv 2025.11.02.685864; doi: https://doi.org/10.1101/2025.11.02.685864 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Bioinformatics Subject Areas All Articles Animal Behavior and Cognition (7619) Biochemistry (17639) Bioengineering (13865) Bioinformatics (41854) Biophysics (21405) Cancer Biology (18544) Cell Biology (25432) Clinical Trials (138) Developmental Biology (13356) Ecology (19863) Epidemiology (2067) Evolutionary Biology (24287) Genetics (15587) Genomics (22465) Immunology (17701) Microbiology (40300) Molecular Biology (17142) Neuroscience (88440) Paleontology (666) Pathology (2825) Pharmacology and Toxicology (4814) Physiology (7633) Plant Biology (15107) Scientific Communication and Education (2042) Synthetic Biology (4285) Systems Biology (9809) Zoology (2268)

Text is read by the "Ask this paper" AI Q&A widget below. Extraction quality varies by source — PMC NXML preserves structure cleanly, OA-HTML may include some navigation residue, and OA-PDF can have broken hyphenation. The publisher copy (via DOI) is the canonical version.

My notes (saved in your browser only)

Ask this paper AI returns verbatim quotes from the full text · source: preprint-html

Answers must be backed by verbatim quotes from this paper's full text. Hallucinated quotes are dropped automatically; if no verbatim passage answers the question, we say so. How this works

Citation neighborhood (no data yet)

We don't have any in-corpus citations linked to this paper yet. This is a recent paper (2025) — citers typically take a year or two to land, and the OpenAlex reference graph may still be filling in.

Source provenance

europepmc
last seen: 2026-05-20T01:45:00.602351+00:00
unpaywall
last seen: 2026-05-21T05:10:58.409756+00:00
License: CC-BY-4.0