Inverted Repeats in Viral Genomes

preprint OA: closed CC-BY-NC-ND-4.0
📄 Open PDF Full text JSON View at publisher
AI-generated deep summary by claude@2026-07, 2026-07-03 · read from full text

This paper studies the prevalence and genomic characteristics of inverted repeats (IRs) in viral DNA, using the Biological Language Modeling Toolkit (BLMT) to analyze 14,000 viral genomes from NCBI. By concatenating each viral genome with its reverse complement and using augmented suffix-array methods to detect repeat pairs that appear in both halves, the authors identify over 19 million IRs longer than 20 bases, including hundreds of very long IRs, and quantify IR density, lengths, percentile positions, and large terminal repeats across viral hosts. A key caveat they highlight is that imperfect IRs with mismatches can be difficult to detect accurately, motivating their approach. The paper does not explicitly discuss endometriosis or adenomyosis; it was included in the corpus via a keyword match in the upstream search index.

Read from the paper's body, not the abstract. Not a substitute for reading the paper. No clinical advice. How this works

Abstract

An inverted repeat (IR) in DNA is a sequence of nucleotides that is followed by its complementary bases but in reverse order, occurring on the same strand (e.g., TCACCGCGGTGA). If the two complementary sequences occur one after the other without other bases between them, they are referred to as DNA palindromes. IRs could form hairpin and cruciform secondary structures, which endanger genomic stability. They are found to be prevalent in viral DNA at origins of replication, and they play a crucial role in various biological processes including gene silencing, duplication, and genomic evolution. IRs have been less explored, which stems from the scarcity of sequence analysis tools allowing accurate detection on large viral genome data. Here, using the Biological Language Modeling Toolkit (BLMT), we analyzed 14 thousand viral genomes for occurrences of IRs, resulting in the identification of over 19 million IRs longer than 20 bases, including 134 IRs that are 2000 bases long, and around 1,300 IRs per virus.
Full text 41,755 characters · extracted from preprint-html · click to expand
Inverted Repeats in Viral Genomes | bioRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-M677548'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search New Results Inverted Repeats in Viral Genomes Madhura Sen , George M. Rivera , Jingxiang Gao , Matthew Shtrahman , View ORCID Profile Madhavi Ganapathiraju doi: https://doi.org/10.1101/2025.11.10.687097 Madhura Sen 1 Vellore Institute of Technology , Vellore, Tamil Nadu, India Find this author on Google Scholar Find this author on PubMed Search for this author on this site George M. Rivera 2 Global Society for Philippine Nurse Researchers , Philippines Find this author on Google Scholar Find this author on PubMed Search for this author on this site Jingxiang Gao 3 Carnegie Mellon University in Qatar Find this author on Google Scholar Find this author on PubMed Search for this author on this site Matthew Shtrahman 4 Department of Neurosciences, School of Medicine, University of California San Diego , USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site Madhavi Ganapathiraju 3 Carnegie Mellon University in Qatar 5 Department of Biomedical Informatics, School of Medicine, University of Pittsburgh , USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Madhavi Ganapathiraju For correspondence: madhavi{at}pitt.edu madhavi{at}cs.cmu.edu Abstract Full Text Info/History Metrics Preview PDF Abstract An inverted repeat (IR) in DNA is a sequence of nucleotides that is followed by its complementary bases but in reverse order, occurring on the same strand (e.g., TCACCGCGGTGA). If the two complementary sequences occur one after the other without other bases between them, they are referred to as DNA palindromes. IRs could form hairpin and cruciform secondary structures, which endanger genomic stability. They are found to be prevalent in viral DNA at origins of replication, and they play a crucial role in various biological processes including gene silencing, duplication, and genomic evolution. IRs have been less explored, which stems from the scarcity of sequence analysis tools allowing accurate detection on large viral genome data. Here, using the Biological Language Modeling Toolkit (BLMT), we analyzed 14 thousand viral genomes for occurrences of IRs, resulting in the identification of over 19 million IRs longer than 20 bases, including 134 IRs that are 2000 bases long, and around 1,300 IRs per virus. Introduction An inverted repeat is a sequence on a DNA strand that is followed by its complement appearing in reverse order on the same strand ( 1 ). The reverse complement may follow immediately after, in which case the pair are referred as a palindrome (e.g., CGAGCTCG) ( 2 ), or be separated by other bases, in which case it is called an inverted repeat (e.g., CGAGtctaCTCG) ( 3 ). Palindromes and inverted repeats where there are a few mismatches in base pairing are known as nearly perfect palindromes and inverted repeats ( 4 ). In this study, we discuss the role of inverted repeats (IR) and their mechanisms related to viral infections, explore existing computational tools for locating such patterns in a gene sequence while introducing a novel tool in Biological Language Modelling Toolkit (BLMT) ( 48 ), and subsequently present analysis on a large-scale of 14 thousand viral genomes determining IR density, lengths, percentile positions, and large terminal repeats across different viral hosts. IRs have a propensity to form hairpin and cruciform structures ( 7 , 8 ), which interfere with DNA replication, DNA damage response, and other genetic mechanisms, induce chromosomal breakage and DNA rearrangements, leading to a variety of diseases, mainly viral infections ( 5 , 6 ). Human SARS-CoV-2 genome was shown to have an abundance of IRs compared to bat CoVs and and deemed to be associated with mutations and recombination events among CoV genomes during the infection phase ( 9 ). Likewise, 2022 monkeypox virus was also noted to differ from the 2018/2019 strain by around 50 variants with a mutation rate of 6 to 12 times higher than anticipated ( 10 ) and linked to IR sequences ( 11 ). IRs are noted to associate with human genetic disorders such as X-linked congenital hypertrichosis syndrome and palindrome-mediated t(3;8) hereditary renal cell carcinoma ( 12 – 20 ). Palindromes were found to be involved in viral packaging, replication, and defense mechanisms ( 21 – 23 ), and are associated with replication initiation sites in viruses such as herpes simplex, varicella-zoster, Epstein-Barr virus (EBV), cytomegalovirus, and baculovirus, and with the recombination process in coronavirus ( 24 – 26 ). Both perfect or nearly perfect palindromes play significant roles in replication initiation and recombination events in viral genomes ( 27 – 30 ) and are associated with diseases including Grave’s disease, multiple sclerosis, and myasthenia gravis. IR-mediated hairpins/cruciforms are hypothesized to modulate replication or recombination; however, for foamy viruses specifically there is no established causal link between IR burden and human neurodegenerative disease, and the evidence is largely observational ( 31 ). Understanding inverted repeats and their role in genetics requires the development of efficient methods to detect them. However, a major caveat arises wherever imperfect inverted repeats are involved, owing to the few mismatches in base pairs at the center or the gap sequence in between the complementary sequences. In this study, we employed the tools available in Biological Language Modelling Toolkit (BLMT) to find IRs using a novel approach. By concatenating each genome with its reverse complement and then using BLMT’s augmented suffix arrays to exhaustively find repeat pairs that occur once in the forward half and once in the reverse-complement half (thereby tagging the two arms of an IR), we were able to capture all the IRs found by the other tools (DetectIR, EMBOSS Palindrome, Palindrome Analyser, and IUPACpal), and also found significantly more IRs uniquely with this method. Data and Methods Data Data for pilot study Genome sequences of five viruses were collected from the NCBI ( https://ftp.ncbi.nlm.nih.gov/genomes/Viruses/ ): ADE virus (ADE), adeno-associated virus (AAV), zika virus, southern bean mosaic virus (SBM), and soybean chlorotic blotch virus (SBC). Data for comprehensive analysis of viral genomes 13 thousand fully sequence viral genomes were collected from the NCBI database, retrieved on February 1st, 2023 ( https://ftp.ncbi.nlm.nih.gov/refseq/release/viral/viral.1.1.genomic.fna.gz ). Methods Suffix array is a data structure that stores suffixes of a sequence starting at every position in a lexicographic order. A suffix array can be augmented with longest common prefix array and rank array which are numerical arrays that allow navigating and pattern mining over suffix array computationally efficient. Biological language modeling toolkit (BLMT) was developed to process genome or proteome sequences into augmented suffix arrays, and to carry out various pattern search operations on it, including finding of k-mer/n-gram counts, location and length of long repeated sequences, and palindromes (i.e., where the reverse complements are not separated by gap sequences). Here, to find inverted repeats, the algorithm uses the observation about the nature of the two halves of IRs. Say, the two halves of the IR are represented as S and R corresponding to the first half sequence and its reverse complement; e.g., if the IR is CGAGtctaCTCG, S would correspond to CGAG and R would correspond to CTCG. If the complement strand of the DNA is considered from its 5’ to 3’, R’s complement would be CGAG, which is the same as S in the original sequence. Thus, if we concatenate the genome and its reverse complement, the long concatenated string will contain these four sequences in order—S, R, S′, and R′—with S′ and R′ appearing in the reverse-complement half. Thus, BLMT, which has a tool to find repeating sequences can be used to locate S and S’ (or R and R’) indicating that the positions corresponding to them are inverse repeats. If the DNA sequence is AAACAGACACGTCTGAAA, the complement strand of the sequence will be TTTCAGACGTGTCTGTTT. The sequence with its reverse complement concatenated to it is AAACAGACACGTCTGAAATTTCAGACGTGTCTGTTT, which will be processed by BLMT. The repetitions found are in bold: AAA CAGA CACG TCTG AAATTT CAGA CGTG TCTG TTT. Here, the repetitions are between the main and complementary sequences, and the presence of an inverted repeat has been confirmed. We first carried out a pilot study to evaluate the performance of the Biological Language Modelling Toolkit (BLMT) by comparing it to the four existing tools and the inverted repeats identified by them. Taking advantage of the efficiency of BLMT inferred from preliminary investigation, we employed BLMT to identify inverted repeats in all existing viral genomes on NCBI. Pilot study to evaluate computing IRs with BLMT Five genomes collected for pilot study were run through the BLMT as described above. For each virus, we took the reverse complement of the entire sequence and concatenated it with the original gene sequence. We call them forback sequences (forward-back) and those were the input sequences for BLMT large repeat finding program (to indirectly find inverted repeats). We also computed the IRs with the computational tools selected for comparison, namely: DetectIR ( 49 ), EMBOSS Palindrome ( 50 ), Palindrome Analyser ( 51 ), IUPACpal ( 52 ). Length of each input sequence (in bp; the length of the genome plus its reverse compliment concatamer): ADE=29254, AAV=9534, Zika=21588, SBC=5416, SBM=8264 To give a consistent scope for all, we evaluated them with the same parameters. Parameters set for all programs and all sequences: Maximum length of half-sequence = 7 Minimum length of half-sequence = 100 Maximum length of gap sequence = 100 Number of mismatches allowed = 0 Study of All Viral Genomes Using BLMT For each of the thirteen thousand fully sequence viral genomes from the NCBI database, we concatenated the genome sequence with its complementary sequence read in reverse order, which forms the input to the analysis. Suffix arrays, longest common prefix arrays and rank arrays were constructed for each input using BLMT. With find large repeats program from BLMT, we identified all repeats that are >= 9 bases. If a pair of repeats occurred at positions x 1 , y 1 , we verified whether y 1 occurred after the midpoint of the new sequence, because it would then correspond to the second half of the IR in the main genome. By assessing whether it is a large repeat within the genome or whether it is a repeat between main and complementary strands, it can be determined whether it is an IR and if so, where it appears in the genome. After performing the analysis for all viruses, we perform the same analysis for viruses on different hosts. Using the NCBI Virus Interface, we selected the six most important host categories: bacteria, fungi, humans, invertebrates, land plants, and vertebrates. For each host selected, we downloaded a complete RefSeq release of viral genomes for that host, and we performed the analysis following the same procedures. Results and Discussion Figure 1 presents the total number of inverted repeat sequences found by each tool for each viral genome. It is clear that BLMT has discovered the highest number of sequences for all five viral genomes. The second highest number of sequences were detected by IUPACpal in ADE and AAV and by DetectIR in Zika, SBC and SBM. Download figure Open in new tab Figure 1: Comparison of IR identification by various tools We carried out the analysis of the complete RefSeq release of viral genomes from the NCBI database, which consisted of 14,797 genomes. We excluded files that contain unresolved nucleotide bases (represented as ‘N’ for any nucleotide), where the palindromic nature or complementarity of the two halves is ambiguous. After preserving files whose sequences are represented by A, T, C, and G, we have 13,023 files remaining for analysis. Lengths of inverted repeats We first analyzed the mean and standard deviation of inverted repeats. Overall, the mean of length of inverted repeats is 10.6, with standard deviation 16.9. The distribution of mean lengths are as shown in Figure 2 . The mean length of inverted repeats for the majority of viruses lies between 10 to 12. Table 1 shows the top 10 viruses with highest mean inverted repeat length. View this table: View inline View popup Download powerpoint Table 1: Viruses with highest average length of inverted repeats Download figure Open in new tab Figure 2. Distribution of mean of length of inverted repeats Among these viruses, there are many species from the Densovirinae subfamily, such as Dendrolimus punctatus densovirus and Diatraea saccharalis densovirus . Moreover, species of Parvoviridae such as Goose parvovirus and Muscovy duck parvovirus also have a high average length of inverted repeats. Each parvorirus contains linear, single-stranded DNA genomes with approximately 5 kb in size and encodes non-structural proteins, which are crucial for viral gene expression and replication ( 32 ). In densovirus, the replication stems from inverted terminal repeats present in the viral genome and involves a mechanism known as rolling circle replication similar to parvovirus ( 33 , 34 ). Among these viruses, there are many species from the Herpesviridae subfamily, such as Gallid herpesvirus and Human herpesvirus. The large standard deviation indicates a wide dispersion of the lengths of inverted repeats, implying the existence of very large inverted repeats in these viruses. Some species of Poxviridae are also present, such as Racoon poxvirus. A study on herpes simplex virus 1 (HSV-1) concluded that replication within the genome occurs between the inverted repeats, producing concatemeric molecules reaching lengths equivalent to 10 times the viral genome ( 35 ). The current MPXV genomes were also identified to be replete with IRs ranging from 6754 to 8933, with a frequency of 34.25–45.11 per kbp ( 11 ). Normalized frequency of inverted repeats The number of viruses with normalized frequency of inverted repeats for every 10,000 base pairs is shown in Figure 3 . Download figure Open in new tab Figure 3. Distribution of viruses by number of inverted repeats per 10kb of genome length Overall, the mean normalized frequency is 20.0. Over half of the viruses having fewer than 15 inverted repeats per 10,000 base pairs. Tables 5 and 6 show the 10 viruses that have the lowest and highest normalized IR frequencies. View this table: View inline View popup Download powerpoint Table 5: Viruses with low normalized frequency of inverted repeats View this table: View inline View popup Download powerpoint Table 6: Viruses with high normalized frequency of inverted repeats From Table 5 , we can identify many fungi-related viruses with low IR frequency: Peanut stunt virus, Sclerotinia sclerotiorum mycoreovirus, Mushroom bacilliform virus, and Flammulina velutipes browning virus. Most fungal viruses with double-stranded RNA (dsRNA) have segmented genomes, which means their genetic material is divided into multiple segments, with each segment being enclosed in a different capsid ( 36 ). This structure inhibits the formation of long stretches of IRs since each segment contains specific genetic information ( 37 ). Also, fungal viruses employ replication strategies such as RNA-dependent RNA polymerase (RdRp)-mediated replication and reverse transcription that may not heavily depend on inverted repeats for effective viral replication, resulting in a lower prevalence of such sequences in their genomes ( 38 , 39 ). Among viruses with high normalized IR frequencies, bacteriophage and poxvirus species commonly appear. Short inverted terminal repeats were discovered in small Bacillus bacteriophage genomes. The study investigated four phages (phi 15, Nf, M2Y, and GA-1) and determined the following terminal repeats, respectively: 5’A-A-A-G-T-A, 5’ A-A-A-G-T-A-A-G for Nf and M2Y, and 5’ A-A-A-T-A-G-A ( 40 ). Similarly, poxviruses contain ITRs with lengths ranging from 1 kb to >17 kb ( 41 ). IR positions within the Genome We analyzed the positions at which inverted repeats start and end in all viral genomes, results of which are shown in Figure 4 . Overall, inverted repeats seem to be more common in the 3’ half of viral genomes compared to 5’ half; there are fewer inverted repeats at the start or end of the viral genome compared to the middle. Download figure Open in new tab Figure 4. Distribution of percentile position of inverted repeats within viral genomes Correlation between inverted repeats and viral hosts We investigated the correlation between inverted repeats in viruses and the types of hosts they infect. After data collection and filtering, we have 4,567 virus genomes with bacteria hosts, 494 virus genomes with fungi hosts, 826 virus genomes with human hosts, 2,100 virus genomes with invertebrate hosts, 1,200 virus genomes with land plant hosts, and 2,676 virus genomes with vertebrate hosts. We computed the mean lengths and normalized frequencies of inverted repeats for viral genomes in each of the categories, results of which are shown in Figure 5 . The average inverted repeat length is similar for all virus categories, with a value between 10 and 11. However, the standard deviation for inverted repeat lengths is especially high for viruses with human, invertebrate, and vertebrate hosts, while it is very low for viruses with bacteria, fungi, and land plant hosts. We have listed the 10 human viruses with the highest average lengths in Table 7 , and highest frequencies in Table 8 . View this table: View inline View popup Table 7: Human viruses with high average length of inverted repeats View this table: View inline View popup Download powerpoint Table 8: Human viruses with high normalized frequency of inverted repeats View this table: View inline View popup Table 10: Table of viruses with long inverted terminal repeats Download figure Open in new tab Figure 5: Mean and standard deviation of inverted repeat lengths for different virus categories Besides commonly occurring species of herpesvirus, direct and inverted repeats in cycloviruses were determined within their putative intergenic region in the genome ( 42 ). On the contrary, the IRs found in Puumala hantavirus were found in the 3’-noncoding region of the S segment. These IRs were found to play a role in recombination events that resulted in the deletion of the sequences responsible for forming hairpin structures ( 43 ). The complete genome of Smallpox variola was also analyzed in other studies and found to contain 725 bp of ITRs with three 69-bp direct repeats. These terminal regions have numerous unique proteins with the potential to enhance the ability of the variola virus to spread and increase its virulence, specifically in humans ( 44 ). Akhmeta virus (AKMV), however, is a novel species of orthopoxvirus (OPXV). A genomic analysis demonstrated that AKMV also shared similarity with OPXV, which has approximately 6 kb sequence in the terminal region ( 45 ). Similar to the studies presented, tandem repeats were observed in Molluscum contagiosum virus, with the proponents associating the presence of such patterns with viral replication ( 46 ). Meanwhile, two identical ITR regions were shown in the Orf virus and Bovine Papular Stomatitis virus genomes ( 47 ). Identifying Inverted Terminal Repeats (ITRs) We also identified Inverted Terminal Repeats (ITRs) by filtering out all inverted repeats that do not occur at the start or end of genomic sequences. Using a Python script, we identified ITRs in 295 viral genomes. We listed the viruses with the longest inverted terminal repeats below: Conclusions It can be inferred from the pilot study that BLMT consistently obtained the highest total number of inverted repeats in all of the sample viruses in comparison to EMBOSS, IUPACpal, detectIR, and DNA Analyzer. On the other hand, EMBOSS, DNA Analyzer, and IUPACpal almost equally identified the same number of inverted repeats. The latter is only a few sequences above EMBOSS in terms of the number of identified inverted repeats. DetectIR, however, found no IRs common to other tools in all of the viruses, which suggests that it completely failed to detect other existing IRs within the genomes. Based on the data, BLMT showed the greatest number of unique sequences. This proves that BLMT is a rather novel bioinformatics tool that can challenge and surpass existing inverted repeat detection tools in terms of accurately identifying imperfect inverted repeats. Our analysis also provides a rigorous analysis of inverted repeats (IRs) in viral genomes, shedding light on their structural characteristics and implications for viral biology. By examining statistical parameters such as the mean and standard deviation of IR lengths, the normalized frequency of IRs, the positions of IRs within genomes, and their correlation with viral hosts, we have achieved a comprehensive understanding of IRs in viral genomes. The analysis of mean IR lengths revealed an average range of 10 to 11 across the examined viruses. Notably, variations in IR lengths were observed among different virus categories, with viruses infecting human, invertebrate, and vertebrate hosts displaying higher standard deviations, indicating greater heterogeneity in IR lengths within these host groups. Conversely, viruses infecting bacteria, fungi, and land plants exhibited lower standard deviations, suggesting a more consistent pattern of IR lengths. The examination of the normalized frequency of IRs provides insights on their prevalence in viral genomes. Bacteria-hosted viruses exhibited a higher frequency of IRs, while viruses infecting fungi and land plants displayed relatively lower frequencies. Remarkably, viruses infecting human, invertebrate, and vertebrate hosts demonstrated a similar mean and standard deviation of normalized IR frequencies, suggesting potential evolutionary similarities and functional implications among these host species. The analysis of IR positions within viral genomes revealed a preference for their occurrence in the latter half of the genome, with a concentration towards the middle region rather than the start or end. This positional distribution highlights the potential functional significance of IRs in viral gene expression and replication processes. Furthermore, our identification of specific viruses with notable characteristics, such as high average IR lengths and extended inverted terminal repeats (ITRs), provides further insights into the genomic structures of these viral species. Species of herpesviruses, densoviruses, and parvoviruses exhibited significant variations in IR lengths, highlighting the diversity within these viral families. Bacteriophages and poxviruses also demonstrated considerable lengths of ITRs, suggesting their importance in viral replication and genetic stability. Our findings contribute to the understanding of viral genome organization and the prevalence of IRs across different virus categories. The insights gained have implications for viral replication, evolution, and host-virus interactions. The comprehensive analysis of IRs in viral genomes enhances our knowledge of viral biology and provides potential avenues for the development of targeted antiviral strategies. In summary, this study highlights the significance of IRs as crucial genomic features in viruses and underscores their diverse characteristics within different virus categories. The examination of statistical parameters and their correlations with viral hosts has advanced our understanding of IRs and their potential functional roles in viral genomes. Further investigations into the functional significance of IRs and their implications for viral pathogenesis will deepen our knowledge of viral biology and inform the development of novel therapeutic approaches. Author Contributions MKG and MSh conceptualized the problem to analyze palindromes in viral genomes. MKG had previously developed the Biological Language Modeling Toolkit, and here developed additional features for viral palindrome identification. MSe, JG and GMR were undergraduate students in three different geographic locations (India, Qatar, Philippines) when this work carried out; MSe and JG received exclusively remote guidance (video calls) from MKG. JG carried out literature reviews. MSe carried out a comparison of different tools with inputs from GMR and MKG. JG carried out post-processing of results on viral families. Manuscript has read and approved by all authors. References 1. ↵ Warburton PE , Giordano J , Cheung F , Gelfand Y , Benson G. Inverted repeat structure of the human genome: the X-chromosome contains a preponderance of large, highly homologous inverted repeats that contain testes genes . Genome Res . 2004 Oct ; 14 ( 10A ): 1861 – 9 . OpenUrl Abstract / FREE Full Text 2. ↵ Anjana R , Shankar M , Vaishnavi MK , Sekar K. A method to find palindromes in nucleic acid sequences . Bioinformation . 2013 ; 9 ( 5 ): 255 – 8 . OpenUrl PubMed 3. ↵ Walker J , Raply R. Molecular Biology and Biotechnology. 5th ed. The Royal Society of Chemistry ; 2009 . 4. ↵ Ganapathiraju MK , Subramanian S , Chaparala S , Karunakaran KB . A reference catalog of DNA palindromes in the human genome and their variations in 1000 Genomes . Hum Genome Var . 2020 Nov 20; 7 ( 1 ): 40 . OpenUrl PubMed 5. ↵ Bowater RP , Bohálová N , Brázda V. Interaction of Proteins with Inverted Repeats and Cruciform Structures in Nucleic Acids . Int J Mol Sci . 2022 May 31; 23 ( 11 ): 6171 . OpenUrl PubMed 6. ↵ Ait Saada A , Guo W , Costa AB , Yang J , Wang J , Lobachev KS . Widely spaced and divergent inverted repeats become a potent source of chromosomal rearrangements in long single-stranded DNA regions . Nucleic Acids Res . 2023 Mar 15; 7. ↵ Voineagu I , Narayanan V , Lobachev KS , Mirkin SM . Replication stalling at unstable inverted repeats: Interplay between DNA hairpins and fork stabilizing proteins . Proceedings of the National Academy of Sciences . 2008 Jul 22; 105 ( 29 ): 9936 – 41 . OpenUrl Abstract / FREE Full Text 8. ↵ Lilley DM . The inverted repeat as a recognizable structural feature in supercoiled DNA molecules . Proc Natl Acad Sci U S A . 1980 Nov ; 77 ( 11 ): 6468 – 72 . OpenUrl Abstract / FREE Full Text 9. ↵ Yin C , Yau SST . Inverted repeats in coronavirus SARS-CoV-2 genome manifest the evolution events . J Theor Biol . 2021 Dec 7; 530 : 110885 . OpenUrl CrossRef PubMed 10. ↵ Isidro J , Borges V , Pinto M , Sobral D , Santos JD , Nunes A , et al. Phylogenomic characterization and signs of microevolution in the 2022 multi-country outbreak of monkeypox virus . Nat Med . 2022 Aug ; 28 ( 8 ): 1569 – 72 . OpenUrl CrossRef PubMed 11. ↵ Dobrovolná M , Brázda V , Warner EF , Bidula S. Inverted repeats in the monkeypox virus genome are hot spots for mutation . J Med Virol . 2023 Jan ; 95 ( 1 ): e28322 . OpenUrl CrossRef PubMed 12. ↵ Bissler JJ . DNA inverted repeats and human disease . Front Biosci . 1998 Mar 27; 3 : d408 – 18 . OpenUrl PubMed 13. Svetec Miklenić M , Svetec IK . Palindromes in DNA-A Risk for Genome Stability and Implications in Cancer . Int J Mol Sci . 2021 Mar 11; 22 ( 6 ). 14. Kato T , Kurahashi H , Emanuel BS . Chromosomal translocations and palindromic AT-rich repeats . Curr Opin Genet Dev . 2012 Jun ; 22 ( 3 ): 221 – 8 . OpenUrl CrossRef PubMed 15. Zhu H , Shang D , Sun M , Choi S , Liu Q , Hao J , et al. X-linked congenital hypertrichosis syndrome is associated with interchromosomal insertions mediated by a human-specific palindrome near SOX3 . Am J Hum Genet . 2011 Jun 10; 88 ( 6 ): 819 – 26 . OpenUrl CrossRef PubMed 16. Kato T , Franconi CP , Sheridan MB , Hacker AM , Inagakai H , Glover TW , et al. Analysis of the t(3;8) of hereditary renal cell carcinoma: a palindrome-mediated translocation . Cancer Genet . 2014 Apr ; 207 ( 4 ): 133 – 40 . OpenUrl CrossRef PubMed 17. Guenthoer J , Diede SJ , Tanaka H , Chai X , Hsu L , Tapscott SJ , et al. Assessment of palindromes as platforms for DNA amplification in breast cancer . Genome Res . 2012 Feb ; 22 ( 2 ): 232 – 45 . OpenUrl Abstract / FREE Full Text 18. Neiman PE , Elsaesser K , Loring G , Kimmel R. Myc oncogene-induced genomic instability: DNA palindromes in bursal lymphomagenesis . PLoS Genet . 2008 Jul 18; 4 ( 7 ): e1000132 . OpenUrl CrossRef PubMed 19. Tanaka H , Cao Y , Bergstrom DA , Kooperberg C , Tapscott SJ , Yao MC . Intrastrand annealing leads to the formation of a large DNA palindrome and determines the boundaries of genomic amplification in human cancer . Mol Cell Biol . 2007 Mar ; 27 ( 6 ): 1993 – 2002 . OpenUrl Abstract / FREE Full Text 20. ↵ Brazda V , Fojta M , Bowater RP . Structures and stability of simple DNA repeats from bacteria . Biochem J . 2020 Jan 31; 477 ( 2 ): 325 – 39 . OpenUrl CrossRef PubMed 21. ↵ Dirac AMG , Huthoff H , Kjems J , Berkhout B. Requirements for RNA heterodimerization of the human immunodeficiency virus type 1 (HIV-1) and HIV-2 genomes . J Gen Virol . 2002 Oct ; 83 ( Pt 10 ): 2533 – 42 . OpenUrl PubMed Web of Science 22. Giedroc DP , Theimer CA , Nixon PL . Structure, stability and function of RNA pseudoknots involved in stimulating ribosomal frameshifting . J Mol Biol . 2000 Apr 28; 298 ( 2 ): 167 – 85 . OpenUrl CrossRef PubMed Web of Science 23. ↵ Karlin S , Burge C , Campbell AM . Statistical analyses of counts and distributions of restriction sites in DNA sequences . Nucleic Acids Res . 1992 Mar 25; 20 ( 6 ): 1363 – 70 . OpenUrl CrossRef PubMed Web of Science 24. ↵ Leung MY , Choi KP , Xia A , Chen LHY . Nonrandom clusters of palindromes in herpesvirus genomes . J Comput Biol . 2005 Apr ; 12 ( 3 ): 331 – 54 . OpenUrl CrossRef PubMed Web of Science 25. Marra MA , Jones SJM , Astell CR , Holt RA , Brooks-Wilson A , Butterfield YSN , et al. The Genome sequence of the SARS-associated coronavirus . Science . 2003 May 30; 300 ( 5624 ): 1399 – 404 . OpenUrl Abstract / FREE Full Text 26. ↵ Elmenofy WH , Jehle JA . Possible functional co-operation of palindromes hr3 and hr4 in the genome of Cydia pomonella granulovirus affects viral replication capacity . J Gen Virol . 2015 Sep ; 96 ( 9 ): 2888 – 97 . OpenUrl PubMed 27. ↵ Cheung AK . Porcine circovirus: transcription and DNA replication . Virus Res . 2012 Mar ; 164 ( 1– 2 ): 46 – 53 . OpenUrl CrossRef PubMed 28. Zhen S , Hua L , Liu YH , Gao LC , Fu J , Wan DY , et al. Harnessing the clustered regularly interspaced short palindromic repeat (CRISPR)/CRISPR-associated Cas9 system to disrupt the hepatitis B virus . Gene Ther . 2015 May ; 22 ( 5 ): 404 – 12 . OpenUrl PubMed 29. Delviks-Frankenberry K , Galli A , Nikolaitchik O , Mens H , Pathak VK , Hu WS . Mechanisms and factors that influence high frequency retroviral recombination . Viruses . 2011 Sep ; 3 ( 9 ): 1650 – 80 . OpenUrl CrossRef PubMed Web of Science 30. ↵ Johnson EM , Wortman MJ , Dagdanova A V , Lundberg PS , Daniel DC . Polyomavirus JC in the context of immunosuppression: a series of adaptive, DNA replication-driven recombination events in the development of progressive multifocal leukoencephalopathy . Clin Dev Immunol . 2013 ; 2013 : 197807 . OpenUrl CrossRef PubMed 31. ↵ Meiering CD , Linial ML . Historical perspective of foamy virus epidemiology and infection . Clin Microbiol Rev . 2001 Jan ; 14 ( 1 ): 165 – 76 . OpenUrl Abstract / FREE Full Text 32. ↵ Martynova EU , Schal C , Mukha D V. Effects of recombination on densovirus phylogeny . Arch Virol . 2016 Jan ; 161 ( 1 ): 63 – 75 . OpenUrl PubMed 33. ↵ Yang B , Zhang J , Cai D , Li D , Chen W , Jiang H , et al. Biochemical characterization of Periplaneta fuliginosa densovirus non-structural protein NS1 . Biochem Biophys Res Commun . 2006 Apr 21; 342 ( 4 ): 1188 – 96 . OpenUrl CrossRef PubMed Web of Science 34. ↵ Kapoor A , Simmonds P , Lipkin WI . Discovery and characterization of mammalian endogenous parvoviruses . J Virol . 2010 Dec ; 84 ( 24 ): 12628 – 35 . OpenUrl Abstract / FREE Full Text 35. ↵ Mahiet C , Ergani A , Huot N , Alende N , Azough A , Salvaire F , et al. Structural variability of the herpes simplex virus 1 genome in vitro and in vivo . J Virol . 2012 Aug ; 86 ( 16 ): 8592 – 601 . OpenUrl Abstract / FREE Full Text 36. ↵ Kielian M , Mettenleiter TC , Roossinck MJ Mata CP , Rodríguez JM , Suzuki N , Castón JR. Chapter Six - Structure and assembly of doublestranded RNA mycoviruses . In: Kielian M , Mettenleiter TC , Roossinck MJ , editors. Advances in Virus Research [Internet] . Academic Press ; 2020 . p. 213 – 47 . Available from: https://www.sciencedirect.com/science/article/pii/S0065352720300373 37. ↵ Sato Y , Castón JR , Suzuki N. The biological attributes, genome architecture and packaging of diverse multi-component fungal viruses . Curr Opin Virol . 2018 Dec ; 33 : 55 – 65 . OpenUrl CrossRef PubMed 38. ↵ Liu YC , Kuo RL , Lin JY , Huang PN , Huang Y , Liu H , et al. Cytoplasmic viral RNA-dependent RNA polymerase disrupts the intracellular splicing machinery by entering the nucleus and interfering with Prp8 . PLoS Pathog . 2014 Jun ; 10 ( 6 ): e1004199 . OpenUrl CrossRef PubMed 39. ↵ Hough B , Steenkamp E , Wingfield B , Read D. Fungal Viruses Unveiled: A Comprehensive Review of Mycoviruses . Viruses . 2023 May 19; 15 ( 5 ). 40. ↵ Yoshikawa H , Ito J. Terminal proteins and short inverted terminal repeats of the small Bacillus bacteriophage genomes . Proc Natl Acad Sci U S A . 1981 Apr ; 78 ( 4 ): 2596 – 600 . OpenUrl Abstract / FREE Full Text 41. ↵ Brennan G , Stoian AMM , Yu H , Rahman MJ , Banerjee S , Stroup JN , et al. Molecular Mechanisms of Poxvirus Evolution . mBio . 2023 Feb 28; 14 ( 1 ): e0152622 . OpenUrl CrossRef PubMed 42. ↵ Todd D , Weston JH , Soike D , Smyth JA . Genome sequence determinations and analyses of novel circoviruses from goose and pigeon . Virology . 2001 Aug 1; 286 ( 2 ): 354 – 62 . OpenUrl CrossRef PubMed 43. ↵ Plyusnina A , Plyusnin A. Recombinant Tula hantavirus shows reduced fitness but is able to survive in the presence of a parental virus: analysis of consecutive passages in a cell culture . Virol J . 2005 Feb 22; 2 : 12 . OpenUrl PubMed 44. ↵ Massung RF , Liu LI , Qi J , Knight JC , Yuran TE , Kerlavage AR , et al. Analysis of the complete genome of smallpox variola major virus strain Bangladesh-1975 . Virology . 1994 Jun ; 201 ( 2 ): 215 – 40 . OpenUrl CrossRef PubMed Web of Science 45. ↵ Gao J , Gigante C , Khmaladze E , Liu P , Tang S , Wilkins K , et al. Genome Sequences of Akhmeta Virus, an Early Divergent Old World Orthopoxvirus . Viruses . 2018 May 12; 10 ( 5 ). 46. ↵ Bugert JJ , Darai G. Stability of molluscum contagiosum virus DNA among 184 patient isolates: evidence for variability of sequences in the terminal inverted repeats . J Med Virol . 1991 Mar ; 33 ( 3 ): 211 – 7 . OpenUrl CrossRef PubMed Web of Science 47. ↵ Delhon G , Tulman ER , Afonso CL , Lu Z , de la Concha-Bermejillo A , Lehmkuhl HD , et al. Genomes of the parapoxviruses ORF virus and bovine papular stomatitis virus . J Virol . 2004 Jan ; 78 ( 1 ): 168 – 77 . OpenUrl Abstract / FREE Full Text 48. ↵ Ganapathiraju M , Manoharan V , Klein-Seetharaman J. BLMT: statistical sequence analysis using N-grams . Appl Bioinformatics . 2004 ; 3 ( 2–3 ): 193 – 200 . OpenUrl CrossRef PubMed 49. ↵ Ye C , Ji G , Li L , Liang C. detectIR: a novel program for detecting perfect and imperfect inverted repeats using complex numbers and vector calculation . PloS one . 2014 Nov 19; 9 ( 11 ): e113349 . OpenUrl CrossRef PubMed 50. ↵ Rice P , Longden I , Bleasby A. EMBOSS: the European molecular biology open software suite . Trends in genetics . 2000 Jun 1; 16 ( 6 ): 276 – 7 . OpenUrl CrossRef PubMed Web of Science 51. ↵ Brázda V , Kolomazník J , Lýsek J , Hároníková L , Coufal J , Št’astný J. Palindrome analyser–a new web-based server for predicting and evaluating inverted repeats in nucleotide sequences . Biochemical and biophysical research communications . 2016 Sep 30; 478 ( 4 ): 1739 – 45 . OpenUrl CrossRef PubMed 52. ↵ Alamro H , Alzamel M , Iliopoulos CS , Pissis SP , Watts S. IUPACpal: efficient identification of inverted repeats in IUPAC-encoded DNA sequences . BMC bioinformatics . 2021 Dec ; 22 ( 1 ): 1 – 2 . OpenUrl CrossRef PubMed View the discussion thread. Back to top Previous Next Posted November 12, 2025. Download PDF Email Thank you for your interest in spreading the word about bioRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following Inverted Repeats in Viral Genomes Message Subject (Your Name) has forwarded a page to you from bioRxiv Message Body (Your Name) thought you would like to see this page from the bioRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share Inverted Repeats in Viral Genomes Madhura Sen , George M. Rivera , Jingxiang Gao , Matthew Shtrahman , Madhavi Ganapathiraju bioRxiv 2025.11.10.687097; doi: https://doi.org/10.1101/2025.11.10.687097 Share This Article: Copy Citation Tools Inverted Repeats in Viral Genomes Madhura Sen , George M. Rivera , Jingxiang Gao , Matthew Shtrahman , Madhavi Ganapathiraju bioRxiv 2025.11.10.687097; doi: https://doi.org/10.1101/2025.11.10.687097 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Bioinformatics Subject Areas All Articles Animal Behavior and Cognition (7633) Biochemistry (17681) Bioengineering (13890) Bioinformatics (41930) Biophysics (21446) Cancer Biology (18586) Cell Biology (25493) Clinical Trials (138) Developmental Biology (13374) Ecology (19897) Epidemiology (2067) Evolutionary Biology (24308) Genetics (15607) Genomics (22498) Immunology (17736) Microbiology (40385) Molecular Biology (17175) Neuroscience (88584) Paleontology (666) Pathology (2831) Pharmacology and Toxicology (4823) Physiology (7641) Plant Biology (15149) Scientific Communication and Education (2045) Synthetic Biology (4293) Systems Biology (9823) Zoology (2271)

Text is read by the "Ask this paper" AI Q&A widget below. Extraction quality varies by source — PMC NXML preserves structure cleanly, OA-HTML may include some navigation residue, and OA-PDF can have broken hyphenation. The publisher copy (via DOI) is the canonical version.

My notes (saved in your browser only)

Ask this paper AI returns verbatim quotes from the full text · source: preprint-html

Answers must be backed by verbatim quotes from this paper's full text. Hallucinated quotes are dropped automatically; if no verbatim passage answers the question, we say so. How this works

Citation neighborhood (no data yet)

We don't have any in-corpus citations linked to this paper yet. This is a recent paper (2025) — citers typically take a year or two to land, and the OpenAlex reference graph may still be filling in.

Source provenance

europepmc
last seen: 2026-05-20T01:45:00.602351+00:00
unpaywall
last seen: 2026-05-27T02:00:06.600101+00:00
License: CC-BY-NC-ND-4.0