Full text
37,806 characters
· extracted from
preprint-html
· click to expand
MSIanalyzer: Targeted Nanopore Sequencing Enables Single Nucleotide Resolution Analysis of Microsatellite Instability Diversity | bioRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-M677548'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search New Results MSIanalyzer: Targeted Nanopore Sequencing Enables Single Nucleotide Resolution Analysis of Microsatellite Instability Diversity View ORCID Profile Ting Zhai , Daniel J. Laverty , Zachary D. Nagel doi: https://doi.org/10.1101/2025.06.26.661510 Ting Zhai 1 Department of Environmental Health, Harvard T.H. Chan School of Public Health , Boston, MA 02115 Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Ting Zhai Daniel J. Laverty 1 Department of Environmental Health, Harvard T.H. Chan School of Public Health , Boston, MA 02115 Find this author on Google Scholar Find this author on PubMed Search for this author on this site For correspondence: dlaverty{at}hsph.harvard.edu znagel{at}hsph.harvard.edu Zachary D. Nagel 1 Department of Environmental Health, Harvard T.H. Chan School of Public Health , Boston, MA 02115 Find this author on Google Scholar Find this author on PubMed Search for this author on this site For correspondence: dlaverty{at}hsph.harvard.edu znagel{at}hsph.harvard.edu Abstract Full Text Info/History Metrics Preview PDF Abstract We present a targeted sequencing-based pipeline that profiles microsatellite instability (MSI) at single-nucleotide resolution. Targeted amplicons from the five widely studied Bethesda panel microsatellite loci were sequenced using Oxford Nanopore Technology in two microsatellite unstable colorectal cancer cell lines (HCT15, HCT116), two microsatellite stable cancer cell lines (TK6, U2OS), and two peripheral blood mononuclear cell samples from healthy donors. An anchor-extension algorithm was developed to capture repeat motifs while allowing interruptions, using a threshold informed by platform-specific error. Cluster-aware Dirichlet-multinomial and beta-binomial tests were applied for between-sample comparisons while accounting for read-level clustering within samples. The algorithm revealed distinct repeat profiles in HCT15 and HCT116 compared to other cell types and uncovered allelic diversity across samples at different MSI loci. Our approach complements existing short tandem repeat callers by preserving read-level diversity and delivering targeted, quantitative MSI calls with potential applications in mechanistic research and clinical assay development. Introduction Microsatellites are highly polymorphic repetitive DNA sequences that exhibit elevated mutation rates 1 . Their tandem repeat structure renders them particularly prone to insertions and deletions during DNA replication, leading to microsatellite instability (MSI), a hallmark of mismatch repair (MMR) deficiency 2 , 3 . Conventional MSI characterization relies on PCR amplification followed by electrophoretic sizing. However, this method provides only average amplicon length estimates from bulk DNA, lacking per-base resolution to identify where and what indels exist, and offers no sequence-level information. Traditional short-read sequencing also faces limitations in accurately resolving MSI regions due to their repetitive nature, resulting in insufficient coverage and limited genomic context. Recent advances in long-read sequencing technologies have opened new opportunities for high-resolution MSI detection and characterization. Oxford Nanopore Technologies (ONT) sequencing has demonstrated utility in diverse applications, including telomere length and DNA modification profiling 4 – 6 . However, ONT’s relatively high error rate (∼5–15%) complicates low-coverage discovery, and most existing error-correction tools are optimized for genome assembly, producing a consensus sequence that collapses intra-sample heterogeneity that may be biologically relevant 7 . While recent efforts in tandem-repeat genotyping provide robust frameworks for population-level analysis, they lack the depth and resolution required to dissect intra-sample MSI variability 8 , 9 . Thus, existing tools have yet to fully leverage deep targeted sequencing to explore MSI heterogeneity. To address this gap, we present MSIanalyzer, a targeted MSI assay that utilizes ONT to capture and characterize MSI variation at the level of individual sequencing reads. Compared to existing short tandem repeat (STR) callers, our approach provides per-read microsatellite context, applies cluster-aware statistical models, and generates interpretable indel visualizations, enabling quantitative, single-nucleotide resolution analysis of MSI complexity that underlies tumor evolution and susceptibility to diseases associated with repeat instability. Methods Sample collection and culture methods We analyzed six samples, including two peripheral blood mononuclear cell (PBMC) samples from anonymous healthy donors, a lymphoblast cell line - TK6, an osteosarcoma cell line - U2OS, two colorectal cancer (CRC) cell lines with known MSI caused by mismatch repair (MMR) deficiencies - HCT15 (MSH6-deficient) and HCT116 (MLH1-deficient). U2OS cells were purchased from ATCC, HCT15 and HCT116 were purchased from the DCDT tumor repository ( https://dtp.cancer.gov/repositories/dctdtumorrepository/default.htm ) at the National Cancer Institute. TK6 cells were purchased from the TK6 Consortium ( http://www.nihs.go.jp/dgm/tk6.html ). PBMCs were purchased from a commercial vendor (SMEBio). U2OS and HCT116 cells were cultured in DMEM high glucose with pyruvate (ThermoFisher Cat. No. 11995065) with 10% fetal bovine serum (FBS, ThermoFisher Cat. No. 10437-028). HCT15 was cultured in RPMI 1640 (Gibco, Cat No. 11875085) with 10% FBS. DNA was isolated directly from quiescent PBMC immediately after recovery from cryopreservation. Sample preparation and sequencing Genomic DNA was isolated from approximately at least 1 million cells per sample using QIAGEN DNeasy Blood & Tissue Kit (Qiagen). Quantity and purity were evaluated by spectrophotometry using a Nanodrop One (ThermoFisher). We targeted five MSI markers from the NCI Bethesda panel: BAT25, BAT26, D2S123, D5S346, and D17S250 10 . PCR primer sequences for each marker were obtained from published sources and are listed in Table 1 . Primer specificity was initially assessed using BLAST against the hg38 reference genome 11 . The exact genomic positions of these markers were further confirmed via UCSC Genome Browser using the STS Marker track for D2S123, D5S346, and D17S250 and the RepeatMasker track for BAT25 and BAT26 12 . View this table: View inline View popup Download powerpoint Table 1. MSI marker details. PCR was performed using 100 ng of input genomic DNA in a 25 μL total reaction volume, consisting of 12.5 µL NEB 2X Q5 Master Mix, 1.25 μL of each 10 µM primer, 1 μL DNA, and 9 μL nuclease-free water. Thermocycling conditions were: initial denaturation at 98°C for 30 s, followed by 35 cycles of 98°C for 10 s, 60°C for 30 s, and 72°C for 90 s. PCR products were purified by Monarch PCR Cleanup kit (New England Biolabs), DNA concentration was determined by Nanodrop, and Nanopore sequencing was performed by Plasmidsaurus using the “Premium PCR” option. Samples were prepared according to the vendor’s protocol for amplicon-based Nanopore sequencing, using input concentrations of 20–200 ng/μL in 10 μL. An average of 5,000 reads per sample was obtained. Read filtering and primer matching Raw FASTQ files underwent initial quality control using FastQC (v0.12.1) to verify the presence of targeted marker sequences and assess overall sequencing quality, including Phred score distributions, base composition, and potential adapter contamination. Reads were subsequently filtered to retain only on-target amplification products. To tolerate errors in Nanopore data, we used a Smith-Waterman local alignment with a similarity threshold (sim ≥ 0.85) 13 , which was calculated below with the Levenshtein distance: Where P is the primer sequence and S the sequence segment being aligned. A read is accepted when both primers align and appear in the correct 5’ to 3’ orientation. Because Nanopore reads may be sequenced from either strand, reads aligning to the reverse strand were reoriented to match the forward-strand convention of the hg38 reference genome. Post-filtering yielded >88% of raw input, with median quality scores around Q30, indicating high-quality, on-target amplification. Repeat detection and motif counting To detect microsatellite repeat units, we implemented a robust anchor-extension algorithm customized for Nanopore data. This approach improves upon traditional longest contiguous repeat counting by accommodating typical Nanopore errors such as substitutions and small indels. The process involved the following steps: 1) Anchor discovery : Reads were scanned using a sliding window of size k, the motif length. Windows containing ≥ 3 consecutive, perfect repeat units were selected as candidate anchors, following established thresholds in STR detection literature 14 , 15 . 2) Bidirectional extension : Each anchor was extended in both directions, unit by unit, allowing up to 1 substitution or ≤ 2-unit indels per step. Extensions terminated once these error thresholds were exceeded. 3) Interruption tracking : Imperfect repeat units encountered during extension were classified as interruptions. For each interruption, we recorded the relative position, type (substitution, insertion, or deletion), and local sequence context. Each read’s output included repeat region boundaries, total pure repeat units, and a catalog of interruptions. 4) Variant signature assignment : Reads containing repeat regions were annotated with variant signatures that summarized both length changes and internal interruptions. Signature categories included: “Perfect[N]”: perfect repeat of N units; “Int[N]”: same length with reference but with ≥ 1 internal interruption; “Exp+X” or “Con–X”: expansion or contraction by X units relative to the reference. Optional tags denoting interruption types and positions were included if observed at a frequency above a default confidence cutoff of 1%, based on expected Nanopore error rates. Reference allele lengths were defined from a matched reference sample or hg38 annotations. Statistical analysis To ensure robust quantification, additional filters were applied to exclude low-support loci (repeat-region read count <5; flanking-region read count <10), removing ∼10% of the post-filtered reads. Thresholds were selected based on empirical performance and prior studies 16 . Our primary objective was to compare repeat unit distributions between samples to identify insertions or deletions at the tandem repeat region. To account for within-sample read correlations, we employed a cluster-aware Dirichlet-multinomial test, which models overdispersed read counts across motif count categories 17 . Specifically, let 𝑋 𝑟s = (𝑥 𝑟s1 , …, 𝑥 𝑟s K ) be read-level motif count vector and thus p be the proportion of motif count for cluster r in sample s , for each motif k we computed: Variances were estimated under a Dirichlet-multinomial model, with the concentration parameter γ̄ derived via maximum likelihood. Multiple testing correction was performed using false discovery rate (FDR) adjustment. To assess read-level indels, we further classified each read as an indel if its repeat unit count deviated from the modal allele(s) observed in a reference sample. For loci with bi- or multi-modal distributions, defined as any repeat count supported by at least 10% total reads post-filtering, the full set of modal alleles was used to define the reference, and any count outside this set was considered an indel. Indel occurrence rates were modelled with a beta-binomial test to account for overdispersion. The underlying indel proportion in each group was modeled as: Where I is the number of indel-supporting reads, N is the total read count, and a = b = 1 for a non-informative Beta(1,1) prior. The posterior distribution of the group-wise difference in indel proportions 𝛥 = 𝑝 1 − 𝑝 2 was used as the test statistic. Posterior p-values were estimated as the proportion of sampled posterior draws with Δ>0 or Δ<0. Additionally, to assess differences in repeat region or amplicon length distributions, we performed non-parametric bootstrap resampling to estimate confidence intervals for length shifts between samples. Visualization Primary visualizations included bar plots of read distributions across repeat units and heatmaps of exact counts across repeat units. To highlight sample-level shifts, we visualized Δ (delta) in motif count proportions between samples. Rare motif lengths in the distribution tails were truncated for visualization but retained in the raw output. To contextualize variant patterns, genome-aligned pileup plots were generated using trimmed hg38 reference sequences (restricted to PCR target regions) and aligned with Minimap2 (v2.28) in ONT mode suitable for Nanopore sequencing data. Pysam (v0.23.3) was used to extract pileup information, including mismatches and indels, for visualization. Representative reads were further visualized by Integrated Genome Viewer (IGV) to provide per-read contexts 18 . Code availability MSIanalyzer is open access on GitHub through: https://github.com/NagelLabHub/MSIanalyzer . A demo is provided on Goolge Colab ( https://colab.research.google.com/drive/13PjP7rVajoGOFAizytyXdfj6Souv2cts?usp=sharing ) for quick analysis with input data. Results Workflow overview Fig. 1 presents the end-to-end MSIanalyzer workflow. Complete analysis of six samples per marker takes approximately 5 minutes, with optional multi-threading available for scalable performance on larger datasets. Download figure Open in new tab Fig. 1. Overview of the MSIanalyzer workflow. Targeted PCR amplicons from microsatellite (MSI) loci are sequenced using Oxford Nanopore Technology (ONT), producing per-sample FASTQ files (Sequencing). A user-supplied JSON manifest links target loci to sequencing data (Linkage). After quality control, reads are matched to their forward and reverse primers with Smith-Waterman alignment (similarity [sim] ≥ 85%), re-oriented to the 5’ to 3’ strand, and off-target reads are discarded (Preprocessing). The core functionality is a robust anchor-extension algorithm, which first detects an anchor of ≥ 3 perfect repeat units, then extends bidirectionally one unit at a time while tolerating ≤ 1 substitution or ≤ 2 indel units per step. Imperfect units encountered during extension are logged in an interruption catalogue (position, edit type, sequence). Final repeat tracts are summarized into variant signatures: Perfect[N], Int[N], Exp+X, or Con−X, with optional annotations (e.g., Con−1 | del@7) if interruptions exceed a defined frequency threshold (see Methods ). Statistical comparisons account for within-sample read clustering using a Dirichlet-multinomial test (repeat unit distributions), beta-binomial model (indel rates), and bootstrap resampling (length shifts). Visual outputs include bar plots, interruption pileups, stacked variant barplots, and summary tables. MSI markers Table 1 lists the five Bethesda-panel loci with their hg38 coordinates, gene contexts, repeat motifs, and primer sequences. BAT25 and BAT26 are A/T mononucleotide tracts, whereas D2S123, D5S346 and D17S250 are AC/TG dinucleotide repeats. Repeat unit distribution across cell lines We compared repeat unit profiles among the two CRC cell lines with known MMR defects and four MMR-proficient controls (two cancer cell lines and two PBMC samples). Fig. 2 and Supplementary Fig. 1 demonstrate clear modal shifts separating MMR-deficient from MMR-intact samples, with p < 0.05 from beta-binomial tests of indel rates compared to the PBMC reference across the five markers ( Supplementary Table 1 ). Most loci displayed high read frequency at multiple repeat units, with 2 to 4 repeat unit lengths supported by at least 10% of reads ( Supplementary Tables 2-6 ). PBMC samples also exhibited allelic variation among themselves, with differential indel rates detected by beta-binomial tests at the dinucleotide loci. Download figure Open in new tab Fig. 2. Repeat unit distribution across samples. Left panels : Smoothed ridge plots showing normalized read frequencies at each repeat unit count for each sample (color-coded). Curves highlight modal shifts and distribution breadth across markers. Right panels : Heatmaps of raw read counts (darker gray = higher counts) across repeat unit bins and samples. Sparse tails are truncated to reduce clutter. Asterisks (*) inside the cells indicate significance (FDR < 0.05) by Dirichlet-multinomial test vs. PBMC8. Asterisks next to sample names indicate a significant indel rate difference (p < 0.05, beta-binomial test) compared to both PBMC7 and PBMC8. At the mononucleotide (T/A) repeat BAT25 , both HCT15 and HCT116 contracted to modes at 28–29 repeat units, with mean reductions of 4 units for HCT15 (95% CI: –4.24 to –3.91) and 5 units for HCT116 (95% CI: –5.33 to –5.01) relative to PBMCs. Dirichlet-multinomial tests confirmed significant under-representation of units 32–35 in both samples (FDR < 0.05), whereas TK6 and U2OS profiles overlapped those of PBMCs. At the mononucleotide (A/T) repeat BAT26 , HCT116 showed a pronounced contraction tail centered at 11–13 units (FDR < 0.05), yielding a mean shift of –8 units (95%CI: –8.13, –7.823). HCT15 contracted more modestly to 15–16 units (mean: −2.1 units, 95% CI: –2.26, –1.90; FDR < 0.05). TK6 and U2OS aligned closely with the PBMC profile. A minor peak at 8 units was detectable in PBMCs and likely represents a rare polymorphism described in prior studies 19 . Its higher frequency in HCT116 suggests instability contributing to contraction to this length. For the dinucleotide (AC/TG) repeat D2S123 , several lines displayed bimodal distributions with peaks at 21 and 27 units. U2OS was almost exclusively centered at 27 units (FDR < 0.05 for units 26–28), whereas HCT116 retained a unimodal 22-unit profile that differed from the modal 21-unit profile in PBMCs (FDR < 0.05 at 21–22 units). HCT15 mirrored the PBMC8 unimodal profile, while TK6 resembled the PBMC7 bimodal profile. At the dinucleotide (TG/AC) repeat D5S346 , HCT15 alone showed a clear expansion peak at 21–22 units, a gain of 6.9 units (95%CI: 6.819, 6.907, FDR < 0.05) compared to PBMC. Other lines retained the 14–16 unit range, although HCT116 showed a secondary peak at 17–18 units (FDR < 0.05). The dinucleotide (GT/CA) repeat D17S250 exhibited the greatest heterogeneity: HCT116 reads spanned 19–40 units, with 14 individual repeat units showing significant differences when compared to PBMC8 (FDR < 0.05), consistent with previous observations of extensive instability at this locus 20 . Modal diversity was also observed across the remaining samples, with PBMC samples being bimodal at 20 and 24 units, TK6 peaked at 19, U2OS at 22, and HCT15 at 20 with a shoulder at 25–26. Per-read variability in indels distribution across cell lines Indel and mismatch rates by genomic position are summarized in Supplementary Fig. 3-7. To visualize per-read variability, we examined BAT26 read stacks in IGV. Fig. 3 contrasts TK6 and HCT116: in TK6, reads showed mostly uniform deletions <10 bp relative to hg38. HCT116 exhibited greater variability, with deletion lengths ranging from 12–18 bp. This further illustrates MSIanalyzer’s ability to capture per-read interruption heterogeneity at single-molecule resolution. Download figure Open in new tab Fig. 3. Read-level indel diversity at BAT26. IGV snapshots show read alignments at BAT26 for TK6 and HCT116. Top: trimmed hg38 reference. Grey: individual reads. Black: deletions in individual reads. Purple numerals: deletion lengths (bp) relative to hg38. Green bases denote the targeted mononucleotide region. Discussion Our results demonstrate that MSIanalyzer accurately captures hallmark MSI patterns in MMR-deficient cell lines and reveals locus-specific differences between MLH1- and MSH6-deficient backgrounds and MMR-proficient controls. Operating directly on sequencing reads, the pipeline preserves per-molecule information, accounts for platform-specific errors without collapsing underlying biological heterogeneity, and evaluates between-sample differences using statistical models that accommodate read clustering. In addition, the ability to visualize read-level interruption patterns enables direct quantification of indel length and frequency. These features make MSIanalyzer a broadly applicable platform not only for MSI profiling but also for other STRs. Several existing tools have advanced the detection of STR variation across platforms. NanoRepeat effectively identifies trinucleotide expansions but is less suited to interruption-rich loci and relies on alignment 21 . Tandem Repeats Finder is widely used for initial STR screening but does not support interruption-aware or quantitative comparisons 22 . NanoSTR is specifically designed for ONT data but lacks per-read diversity reporting 23 . OS-Seq enables high-accuracy, multiplex MSI profiling using custom probe-based capture and primer extension, though it relies on specialized oligonucleotide synthesis and instrumentation, which may limit flexibility for broader applications 24 . MSIanalyzer complements these tools by focusing on preserving intra-sample allelic variation and read-level variation, enabling single-molecule resolution and visualization of MSI patterns. We acknowledge several limitations of our approach. First, as a targeted approach relying on PCR amplification, MSIanalyzer is susceptible to amplification bias and polymerase slippage, potentially distorting allele spectra, particularly at very long homopolymers. Second, this proof-of-concept study includes only six samples, limiting power to detect rare alleles. However, the implementation of a Dirichlet-multinomial test helps address overdispersion. Third, we observed discrepancies in motif counts compared to the hg38 reference genome, likely stemming from persistent platform-specific sequencing artifacts. Without external validation or larger sample cohorts, these cannot be definitively resolved. Nonetheless, within-sample comparisons remain robust, as all samples are subject to the same artifacts, which are mitigated through a robust analytical framework. Despite these limitations, our results reveal locus-specific distinctions between MLH1- and MSH6-deficient CRC cell lines and MMR-proficient controls, supporting that different DNA repair deficiencies produce characteristic MSI signatures 25 . If confirmed in larger cohorts, these patterns could inform tumor classification and enhance biomarker discovery for precision oncology. Future development of MSIanalyzer will focus on benchmarking against complex repeat regions in population-scale datasets and validate the statistical framework under real-world heterogeneity. With continued refinement, MSIanalyzer has the potential to become a general-purpose, low-cost solution for research and clinical MSI typing. Declaration of Interests Z.D.N. received funding for this project from Pfizer Inc., holds a patent titled “Methods and kits for determining DNA repair capacity,” and has received research funding from Intellia Therapeutics, Ensoma, and Agios that is unrelated to this project. Author Contributions Conceptualization and methodology: TZ, DJL, ZDN Investigation: TZ, DJL Software, formal analysis, visualization, writing – original draft: TZ Resources, writing – review and revision: DJL, ZDN Supervision and funding acquisition: ZDN Download figure Open in new tab Supplementary Fig. 1. Repeat unit distribution across samples. Each bar represents the percentage of reads observed at a given repeat unit, grouped by sample and color-coded as indicated in the legend. Download figure Open in new tab Download figure Open in new tab Supplementary Fig. 2. Difference in repeat unit proportions across samples. The difference in read proportion (Δ % Reads) for each sample compared to the PBMC8 reference. Bars show differences in repeat unit frequencies, colored by indel classification: enriched or depleted insertions/deletions, relative to the reference, as defined in the Motif Change key. Gray bars represent reference motif counts (most frequent in PBMC8). Download figure Open in new tab Supplementary Fig. 3. Per-base indel and mismatch pileup profiles at BAT25. Top panel : smoothed frequency of indels (blue) and mismatches (gold) along the locus. Middle panel : hg38 reference base track. Bottom panel : detailed indel visualization showing insertion and deletion types, lengths (in bp), and positions across reads. Download figure Open in new tab Supplementary Fig. 4. Per-base indel and mismatch pileup profiles at BAT26. Top panel : smoothed frequency of indels (blue) and mismatches (gold) along the locus. Middle panel : hg38 reference base track. Bottom panel : detailed indel visualization showing insertion and deletion types, lengths (in bp), and positions across reads. Download figure Open in new tab Supplementary Fig. 5. Per-base indel and mismatch pileup profiles at D2S123. Top panel : smoothed frequency of indels (blue) and mismatches (gold) along the locus. Middle panel : hg38 reference base track. Bottom panel : detailed indel visualization showing insertion and deletion types, lengths (in bp), and positions across reads. Download figure Open in new tab Supplementary Fig. 6. Per-base indel and mismatch pileup profiles at D5S346. Top panel : smoothed frequency of indels (blue) and mismatches (gold) along the locus. Middle panel : hg38 reference base track. Bottom panel : detailed indel visualization showing insertion and deletion types, lengths (in bp), and positions across reads. Download figure Open in new tab Supplementary Fig. 7. Per-base indel and mismatch pileup profiles at D17S250. Top panel : smoothed frequency of indels (blue) and mismatches (gold) along the locus. Middle panel : hg38 reference base track. Bottom panel : detailed indel visualization showing insertion and deletion types, lengths (in bp), and positions across reads. View this table: View inline View popup Download powerpoint Supplementary Table 1. Statistical comparison for indel rate and repeat unit length shifts across samples. View this table: View inline View popup Download powerpoint Supplementary Table 2. Repeat unit distribution and statistical comparisons for BAT25 across six samples. View this table: View inline View popup Download powerpoint Supplementary Table 3. Repeat unit distribution and statistical comparisons for BAT26 across six samples. View this table: View inline View popup Download powerpoint Supplementary Table 4. Repeat unit distribution and statistical comparisons for D2S123 across six samples. View this table: View inline View popup Download powerpoint Supplementary Table 5. Repeat unit distribution and statistical comparisons for D5S346 across six samples. View this table: View inline View popup Download powerpoint Supplementary Table 6. Repeat unit distribution and statistical comparisons for D17S250 across six samples. Acknowledgments This research was partially supported by a sponsored research agreement with Pfizer Inc. ZDN and DJL also acknowledge support from American Cancer Society grant (RSG-22-038-01) and NIH grant (P01CA092584). Funder Information Declared Pfizer Inc American Cancer Society , RSG-22-038-01 National Institutes of Health, https://ror.org/01cwqze88 , P01CA092584 References 1. ↵ Aaltonen , L.A. , et al. Clues to the pathogenesis of familial colorectal cancer . Science 260 , 812 – 816 ( 1993 ). OpenUrl Abstract / FREE Full Text 2. ↵ Ellegren , H . Microsatellites: simple sequences with complex evolution . Nat Rev Genet 5 , 435 – 445 ( 2004 ). OpenUrl CrossRef PubMed Web of Science 3. ↵ Li , G.M . Mechanisms and functions of DNA mismatch repair . Cell Res 18 , 85 – 98 ( 2008 ). OpenUrl CrossRef PubMed Web of Science 4. ↵ Jain , M. , et al. Nanopore sequencing and assembly of a human genome with ultra-long reads . Nat Biotechnol 36 , 338 – 345 ( 2018 ). OpenUrl CrossRef PubMed 5. Kong , Y. , Mead , E.A. & Fang , G . Navigating the pitfalls of mapping DNA and RNA modifications . Nat Rev Genet 24 , 363 – 381 ( 2023 ). OpenUrl CrossRef PubMed 6. ↵ Schmidt , T.T. , et al. High resolution long-read telomere sequencing reveals dynamic mechanisms in aging and cancer . Nat Commun 15 , 5149 ( 2024 ). OpenUrl CrossRef PubMed 7. ↵ Logsdon , G.A. , Vollger , M.R. & Eichler , E.E . Long-read human genome sequencing and its applications . Nat Rev Genet 21 , 597 – 614 ( 2020 ). OpenUrl PubMed 8. ↵ Dolzhenko , E. , et al. Characterization and visualization of tandem repeats at genome scale . Nat Biotechnol 42 , 1606 – 1614 ( 2024 ). OpenUrl CrossRef PubMed 9. ↵ Shi , Y. , et al. Characterization of genome-wide STR variation in 6487 human genomes . Nat Commun 14 , 2092 ( 2023 ). OpenUrl CrossRef PubMed 10. ↵ Boland , C.R. , et al. A National Cancer Institute Workshop on Microsatellite Instability for cancer detection and familial predisposition: development of international criteria for the determination of microsatellite instability in colorectal cancer . Cancer Res 58 , 5248 – 5257 ( 1998 ). OpenUrl Abstract / FREE Full Text 11. ↵ Altschul , S.F. , Gish , W. , Miller , W. , Myers , E.W. & Lipman , D.J . Basic local alignment search tool . J Mol Biol 215 , 403 – 410 ( 1990 ). OpenUrl CrossRef PubMed Web of Science 12. ↵ Perez , G. , et al. The UCSC Genome Browser database: 2025 update . Nucleic Acids Res 53 , D1243 – D1249 ( 2025 ). OpenUrl CrossRef PubMed 13. ↵ Smith , T.F. & Waterman , M.S . Identification of common molecular subsequences . J Mol Biol 147 , 195 – 197 ( 1981 ). OpenUrl CrossRef PubMed Web of Science 14. ↵ Ding , Z. , et al. Estimating telomere length from whole genome sequence data . Nucleic Acids Res 42 , e75 ( 2014 ). OpenUrl CrossRef PubMed 15. ↵ Zhai , T. & Nagel , Z.D. Current Technologies for Measuring or Predicting Telomere Length from Genomic Datasets . in Population Genetics - From DNA to Evolutionary Biology (ed. Behzadi , P. ) ( IntechOpen , Rijeka , 2024 ). 16. ↵ Niu , B. , et al. MSIsensor: microsatellite instability detection using paired tumor-normal sequence data . Bioinformatics 30 , 1015 – 1016 ( 2014 ). OpenUrl CrossRef PubMed 17. ↵ Chu , T. , Wang , Z. , Pe’er , D. & Danko , C.G . Cell type and gene expression deconvolution with BayesPrism enables Bayesian integrative analysis across bulk and single-cell RNA sequencing in oncology . Nat Cancer 3 , 505 – 517 ( 2022 ). OpenUrl CrossRef PubMed 18. ↵ Robinson , J.T. , et al. Integrative genomics viewer . Nat Biotechnol 29 , 24 – 26 ( 2011 ). OpenUrl CrossRef PubMed Web of Science 19. ↵ Pyatt , R. , et al. Polymorphic variation at the BAT-25 and BAT-26 loci in individuals of African origin. Implications for microsatellite instability testing . Am J Pathol 155 , 349 – 353 ( 1999 ). OpenUrl CrossRef PubMed Web of Science 20. ↵ Matsuno , Y. , et al. Replication stress triggers microsatellite destabilization and hypermutation leading to clonal expansion in vitro . Nat Commun 10 , 3925 ( 2019 ). OpenUrl CrossRef PubMed 21. ↵ Fang , L. , et al. Haplotyping SNPs for allele-specific gene editing of the expanded huntingtin allele using long-read sequencing . HGG Adv 4 , 100146 ( 2023 ). OpenUrl PubMed 22. ↵ Benson , G . Tandem repeats finder: a program to analyze DNA sequences . Nucleic Acids Res 27 , 573 – 580 ( 1999 ). OpenUrl CrossRef PubMed Web of Science 23. ↵ Lang , J. , Xu , Z. , Wang , Y. , Sun , J. & Yang , Z . NanoSTR: A method for detection of target short tandem repeats based on nanopore sequencing data . Front Mol Biosci 10 , 1093519 ( 2023 ). OpenUrl CrossRef PubMed 24. ↵ Shin , G. , et al. Profiling diverse sequence tandem repeats in colorectal cancer reveals co-occurrence of microsatellite and chromosomal instability involving Chromosome 8 . Genome Med 13 , 145 ( 2021 ). OpenUrl CrossRef PubMed 25. ↵ Carethers , J.M. , Koi , M. & Tseng-Rogenski , S.S . EMAST is a Form of Microsatellite Instability That is Initiated by Inflammation and Modulates Colorectal Cancer Progression . Genes (Basel ) 6 , 185 – 205 ( 2015 ). OpenUrl PubMed View the discussion thread. Back to top Previous Next Posted June 28, 2025. Download PDF Email Thank you for your interest in spreading the word about bioRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following MSIanalyzer: Targeted Nanopore Sequencing Enables Single Nucleotide Resolution Analysis of Microsatellite Instability Diversity Message Subject (Your Name) has forwarded a page to you from bioRxiv Message Body (Your Name) thought you would like to see this page from the bioRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share MSIanalyzer: Targeted Nanopore Sequencing Enables Single Nucleotide Resolution Analysis of Microsatellite Instability Diversity Ting Zhai , Daniel J. Laverty , Zachary D. Nagel bioRxiv 2025.06.26.661510; doi: https://doi.org/10.1101/2025.06.26.661510 Share This Article: Copy Citation Tools MSIanalyzer: Targeted Nanopore Sequencing Enables Single Nucleotide Resolution Analysis of Microsatellite Instability Diversity Ting Zhai , Daniel J. Laverty , Zachary D. Nagel bioRxiv 2025.06.26.661510; doi: https://doi.org/10.1101/2025.06.26.661510 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Molecular Biology Subject Areas All Articles Animal Behavior and Cognition (7617) Biochemistry (17633) Bioengineering (13856) Bioinformatics (41841) Biophysics (21399) Cancer Biology (18529) Cell Biology (25422) Clinical Trials (138) Developmental Biology (13352) Ecology (19860) Epidemiology (2067) Evolutionary Biology (24281) Genetics (15582) Genomics (22461) Immunology (17700) Microbiology (40293) Molecular Biology (17140) Neuroscience (88413) Paleontology (666) Pathology (2823) Pharmacology and Toxicology (4813) Physiology (7632) Plant Biology (15107) Scientific Communication and Education (2042) Synthetic Biology (4284) Systems Biology (9808) Zoology (2267)
Text is read by the "Ask this paper" AI Q&A widget below.
Extraction quality varies by source — PMC NXML preserves structure
cleanly, OA-HTML may include some navigation residue, and OA-PDF can
have broken hyphenation. The publisher copy
(via DOI)
is the canonical version.