Example of Methylome Analysis with MethylIT using Cancer Datasets

preprint OA: closed CC-BY-NC-ND-4.0
📄 Open PDF Full text JSON View at publisher
⚙ AI-generated deep summary by qwen3.7-flash, 2026-09-24 · read from full text ⓘ

This paper presents a methodological tutorial for applying MethylIT, an R package based on information thermodynamics and signal detection theory, to analyze whole-genome bisulfite sequencing data. The authors demonstrate the software’s installation, dataset retrieval from public repositories, and analytical workflow using breast cancer methylome data specifically restricted to chromosome 13. They emphasize that the method aims to distinguish true methylation regulatory signals from background thermal noise by establishing a reference individual derived from embryonic stem cells or control group averages. This paper does not explicitly discuss endometriosis or adenomyosis; it was included in the corpus via a keyword match in the upstream search index.

Read from the paper's body, not the abstract. Not a substitute for reading the paper. No clinical advice. How this works

Abstract

Methyl-IT, a novel methylome analysis procedure based on information thermodynamics and signal detection was recently released. Methylation analysis involves a signal detection problem, and the method was designed to discriminate methylation regulatory signal from background noise induced by thermal fluctuations. Methyl-IT enhances the resolution of genome methylation behavior to reveal network-associated responses, offering resolution of gene pathway influences not attainable with previous methods. Herein, an example of MethylIT application to the analysis of breast cancer methylomes is presented.
Full text 43,903 characters · extracted from preprint-html · click to expand
Example of Methylome Analysis with MethylIT using Cancer Datasets | bioRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-M677548'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search Confirmatory Results Example of Methylome Analysis with MethylIT using Cancer Datasets View ORCID Profile Robersy Sanchez , View ORCID Profile Sally Mackenzie doi: https://doi.org/10.1101/261982 Robersy Sanchez 1 Department of Biology and Plant Science. Pennsylvania State University , University Park, PA 16802 Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Robersy Sanchez Sally Mackenzie 1 Department of Biology and Plant Science. Pennsylvania State University , University Park, PA 16802 Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Sally Mackenzie Abstract Full Text Info/History Metrics Preview PDF Abstract Methyl-IT, a novel methylome analysis procedure based on information thermodynamics and signal detection was recently released. Methylation analysis involves a signal detection problem, and the method was designed to discriminate methylation regulatory signal from background noise induced by thermal fluctuations. Methyl-IT enhances the resolution of genome methylation behavior to reveal network-associated responses, offering resolution of gene pathway influences not attainable with previous methods. Herein, an example of MethylIT application to the analysis of breast cancer methylomes is presented. 1 MethylIT MethylIT is an R package for methylome analysis based on information thermodynamics and signal detection. The information thermodynamics-based approach is postulated to provide greater sensitivity for resolving true signal from the thermodynamic background within the methylome (Sanchez and Mackenzie 2016). Because the biological signal created within the dynamic methylome environment characteristic of plants is not free from background noise, the approach, designated MethylIT, includes the application of signal detection theory (Greiner, Pfeiffer, and Smith 2000; Carter et al. 2016 ; Harpaz et al. 2013 ; Kruspe et al. 2017 ). A basic requirement for the application of signal detection is a probability distribution of the background noise. Probability distribution, as a Weibull distribution model, can be deduced on a statistical mechanical/thermodynamics basis for DNA methylation induced by thermal fluctuations (Sanchez and Mackenzie 2016). Assuming that this background methylation variation is consistent with a Poisson process, it can be distinguished from variation associated with methylation regulatory machinery, which is non-independent for all genomic regions (Sanchez and Mackenzie 2016). An information-theoretic divergence to express the variation in methylation induced by background thermal fluctuations will follow a Weibull distribution model, provided that it is proportional to minimum energy dissipated per bit of information from methylation change. Herein, we provide an example of MethylIT application to the analysis of breast cancer methylomes. Due to the size of human methylome the current example only covers the analysis of chromosome 13. A full description of MethylIT application of methylome analysis in plants is given in the manuscript ( Sanchez et al. 2018 ). 1.1 Installation of MethylIT To install MethylIT you might need to install the Bioconductor packages: ‘GenomicFeatures’, ‘VariantAnnotation’, ‘ensembldb’, ‘GenomicRanges’, ‘BiocParallel’, ‘biovizBase’, ‘DESeq2’, and ‘genefilter’. Please check that both the R and bioconductor packages are up to date: update.packages(ask = FALSE) source(‘ https://bioconductor.org/biocLite.R ’) biocLite(ask = FALSE) MethylIT can be installed from PSU’s GitLab by typing in an R console: install.packages(‘devtools’) devtools::install_git(‘ https://git.psu.edu/genomath/MethylIT ’) Some possible troubleshooting installation on Ubuntu is given in section S1. Installation on our Windows OS machines was straightforward. 2 Available datasets and reading Methylome datasets of whole-genome bisulfite sequencing (WGBS) are available at Gene Expression Omnibus (GEO DataSets). For the current example, datasets from breast tissues (normal and cancer) and embryonic stem cells will be downloaded from GEO. The data set are downloaded providing the GEO accession numbers for each data set to the function ‘getGEOSuppFiles’ (for details type ?getGEOSuppFiles in the R console). Download figure Open in new tab The file path and name of each downloaded dataset is found in the output variables ‘esc.files’ and ‘cancer.files’. 2.1 Reading datasets Datasets for our example can be read with function ‘readCounts2GRangesList’. To specify the reading of only chromosome 13, we can specify the parameter ‘chromosomes = “Chr13”’. The symbol chromosome 13, in this case “Chr13”, must be consistent with the annotation provided in the given GEO dataset. Each file is wholly read with the setting ‘chromosomes = “Chr13”’ and then the GRanges are built only with chromosome 13, which could be time consuming. However, users working on Linux OS can specify the reading of specific lines from each file by using regular expressions. For example, if only chromosomes 1 and 3 are required, then we can set chromosomes = NULL (default) and ‘chromosome.pattern = “ˆChr[1,3]”’. This will read all the lines in the downloaded files starting with the words “Chr1” or “Chr3”. If we are interested in chromosomes 1 and 2, then we can set ‘chromosome.pattern = “ˆChr[1-2]”’. If all the chromosomes are required, then set chromosomes = NULL and chromosome.pattern = NULL (default). Download figure Open in new tab In the metacolumn of the last GRanges object, mC and uC stand for the methylated and unmethy-lated read counts, respectively. Notice that option ‘remove = TRUE’ remove the decompressed files (default: FALSE, see ?readCounts2GRangesList for more details about this function). 3. The reference individual Any two objects located in a space can be compared if, and only if, there is a reference point (a coordinate system) in the space and a metric. Usually, in our daily 3D experience, our brain automatically sets up the origin of coordinates equal to zero. The differences found in the comparison depend on the reference used to perform the measurements and from the metric system. The space where the objects are located (or the set of objects) together with the metric is called metric space. To evaluate the methylation differences between individuals from control and treatment we introduce a metric in the bidimensional space of methylation levels P i = ( p i , 1 − P i ). Vectors P i provide a measurement of the uncertainty of methylation levels. However, to perform the comparison between the uncertainty of methylation levels from each group of individuals, control ( c ) and treatment ( t ), we should estimate the uncertainty variation with respect to the same individual reference on the mentioned metric space. The reason to measure the uncertainty variation with respect to the same reference resides in that even sibling individuals follow an independent ontogenetic development. This a consequence of the “omnipresent” action of the second law of thermodynamics in living organisms. In the current example, we will create the reference individual by pooling the methylation counts from the embryonic stem cells. It should be noticed that the results are sensitive to the reference used. The statistics mean, median, or sum of the read counts at each cytosine site of some control samples can be used to create a virtual reference sample. It is up to the user whether to apply the ‘row sum’, ‘row mean’ or ‘row median’ of methylated and unmethylated read counts at each cytosine site across individuals: Download figure Open in new tab View this table: View inline View popup Download powerpoint Only direct lab experiments can reveal whether differences detected with distinct references outside the experimental conditions for control and treatment groups are real. The best reference would be estimated using a subset of individuals from control group. Such a reference will contribute to remove the intragroup variation, in control and in treatment groups, induced by environmental changes external to or not controlled by the experimental conditions. Methylation analysis for each cystosine position is frequently performed in the bidimensional space of ( methylated, unmethylated ) read counts. Frequently, Fisher test is applied to a single cytosine position, under the null hypothesis that the proportions p ct = methylated ct /( methylated ct + unmethylated ct ) and p tt = methylated tt /( methylated tt + unmethylated tt ) are the same for control and treatment, respectively. In this case, the implicit reference point for the counts at every cytosine positions is ( methylated = 0, unmethylated = 0), which corresponds to the point P i = (0, 1). In our case, the Hellinger divergence (the metric used, here) of each individual in respect to the reference is the variable to test in place of ( methylated, unmethylated ) read counts or the methylation levels P i = ( p i , 1 − p i ). The use of references is restricted by the thermodynamics basis of the the theory. The current information-thermodynamics based approach is supported on the following postulate: “High changes of Hellinger divergences are less frequent than low changes, provided that the divergence is proportional to the amount of energy required to process one bit of information in methylation system” . The last postulate acknowledges the action of the second law of thermodynamics on the biomolecular methylation system. For the methylation system, it implies that the frequencies of the information divergences between methylation levels must be proportional to a Boltzmann factor (see supplementary information from reference (Sanchez and Mackenzie 2016)). In other words, the frequencies of information divergences values should follow a trend proportional to an exponential decay. If we do not observe such a behaviour, then either the reference is too far from experimental condition or we are dealing with an extreme situation where the methylation machinery in the cell is dysfunctional. The last situation is found, for example, in the silencing mutation at the gene of cytosine-DNA-methyltransferase in Arabidopsis thaliana . Methylation of 5-methylcytosine at CpG dinucleotides is maintained by MET1 in plants. In our current example, the embryonic stem cells reference is far from the breast tissue samples and this could affect the nonlinear fit to a Weibull distribution (see below). To illustrate the effect of the reference on the analysis, a new reference will be built by setting: Download figure Open in new tab The reason for the above replacement is that natural methylation changes (Ref$mC) obey the second law of thermodynamics, and we do not want to arbitrarily change the number of methylated read counts. ‘mC’ carries information linked to the amount of energy expended in the tissue associated with concrete methylation changes. However, ‘uC’ is not linked to any energy expended by the methylation machinery in the cells. In the bidimensional space P i = ( p i , 1 − P i ), reference Ref0 corresponds to the point P i = (1, 0) at each cytosine site i , i.e., the value of methylation level at every cytosine site in reference Ref0 is 1. The analyses with respect to both individual references, Ref and Ref0 , will be performed in the downstream steps. 4 Hellinger divergence estimation To perform the comparison between the uncertainty of methylation levels from each group of individuals, control ( c ) and treatment ( t ), the divergence between the methylation levels of each individual is estimated with respect to the same reference on the metric space formed by the vector set P i = ( p i , 1 − P i ) and the Hellinger divergence H . Basically, the information divergence between the methylation levels of an individual j and reference sample r is estimated according to the Hellinger divergence given by the formula: where and j ϵ { c, t }. This equation for Hellinger divergence is given in reference ( Basu, Mandal, and Pardo 2010 ), but other information theoretical divergences can be used as well. Next, the information divergence for control (Breast_normal) and treatment (Breast_cancer and Breast_metastasis) samples are estimated with respect to the reference virtual individual. A Bayesian correction of counts can be selected or not. In a Bayesian framework, methylated read counts are modeled by a beta-binomial distribution, which accounts for both the biological and sampling variations ( Hebestreit, Dugas, and Klein 2013 ; Robinson et al. 2014 ; Dolzhenko and Smith 2014 ). In our case we adopted the Bayesian approach suggested in reference ( Baldi and Brunak 2001 ) (Chapter 3). In a Bayesian framework with uniform priors, the methylation level can be defined as: p = ( mC + 1) / ( mC + uC + 2). However, the most natural statistical model for replicated BS-seq DNA methylation measurements is beta-binomial (the beta distribution is a prior conjugate of binomial distribution). We consider the parameter p (methylation level) in the binomial distribution as randomly drawn from a beta distribution. The hyper-parameters α and β from the beta-binomial distribution are interpreted as pseudo-counts. The information divergence is estimated here using the function ‘estimateDivergence’: Download figure Open in new tab View this table: View inline View popup Function ‘estimateDivergence’ returns a list of GRanges objects with the four columns of counts, the information divergence, and additional columns: The original matrix of methylated ( c i ) and unmethylated ( t i ) read counts from control ( i = 1) and treatment ( i = 2) samples. “p1” and “p2”: methylation levels for control and treatment, respectively. “bay.TV”: total variation TV = p2 - p1. “TV”: total variation based on simple counts: T V = c 1 / ( c 1 + t 1) c 2 / ( c 2 + t 2). “hdiv”: Hellinger divergence. If Bayesian = TRUE, results are based on the posterior estimations of methylation levels p 1 and p 2. Filtering by coverage is provided at this step which would be used unless previous filtering by coverage had been applied. This is a pairwise filtering. Cytosine sites with ‘coverage’ > ‘min.coverage’ and ‘coverage’ < ‘percentile’ (e.g., 99.9 coverage percentile) in at least one of the samples are preserved. The coverage percentile used is the maximum estimated from both samples: reference and individual. For some GEO datasets only the methylation levels for each cytosine site are provided. In this case, Hellinger divergence can be estimated as given in reference ( Sanchez and Mackenzie 2016 ): 4.1. Histogram and boxplots of divergences estimated in each sample First, the data of interest (Hellinger divergences, “hdiv”) are selected from the GRanges objects: Download figure Open in new tab Next, a single GRanges object is built from the above set of GRanges objects using the function ‘uniqueGRanges’. Notice that the number of cores to use for parallel computation can be specified. Download figure Open in new tab View this table: View inline View popup Now, the Hellinger divergences estimated for each sample are in a single matrix on the metacolumn of the GRanges object and we can proceed to build the histogram and boxplot graphics for these data. Download figure Open in new tab View this table: View inline View popup Download figure Open in new tab Download figure Open in new tab Except for the tail, most of the methylation changes occurred under the area covered by the density curve corresponding to the normal breast tissue. This is theoretically expected. This area is explainable in statistical physical terms and, theoretically, it should fit a Weibull distribution. The tails regions cover the methylation changes that, with high probability, are not induced by thermal fluctuation and are not addressed to stabilize the DNA molecule. These changes are methylation signal. Professor David J. Miller (Department of Electrical Engineering, Penn State) proposed modeling the distribution as a mixed Weibull distribution to simultaneously describe the background methylation noise and the methylation signal (personal communication, January, 2018). This model approach seems to be supported by the above histogram, but it must be studied before being incorporated in a future version of Methyl-IT. 5. Nonlinear fit of Weibull distribution A basic requirement for the application of signal detection is the knowledge of the probability distribution of the background noise. Probability distribution, as a Weibull distribution model, can be deduced on a statistical mechanical/thermodynamics basis for DNA methylation induced by thermal fluctuations (Sanchez and Mackenzie 2016). Assuming that this background methylation variation is consistent with a Poisson process, it can be distinguished from variation associated with methylation regulatory machinery, which is non-independent for all genomic regions (Sanchez and Mackenzie 2016). An information-theoretic divergence to express the variation in methylation induced by background thermal fluctuations will follow a Weibull distribution model, provided that it is proportional to the minimum energy dissipated per bit of information associated with the methylation change. The nonlinear fit to a Weibull distribution model is performed by the function ‘nonlinearFitDist’. Download figure Open in new tab View this table: View inline View popup Cross-validations for the nonlinear regressions (R.Cross.val) were performed as described in reference ( Stevens 2009 ). In addition, Stein’s formula for adjusted R squared (ρ) was used as an estimator of the average cross-validation predictive power ( Stevens 2009 ). The goodness-of-fit of Weibull to the HD0 ( Ref0 ) data is better than to HD ( Ref ): nlms0 View this table: View inline View popup The goodness-of-fit indicators suggest that the fit to Weibull distribution model for Ref0 is better than for Ref . 6 Signal detection The information thermodynamics-based approach is postulated to provide greater sensitivity for resolving true signal from the thermodynamic background within the methylome (Sanchez and Mackenzie 2016). Because the biological signal created within the dynamic methylome environment characteristic of plants is not free from background noise, the approach, designated Methyl-IT, includes the application of signal detection theory ( Greiner, Pfeiffer, and Smith 2000 ; Carter et al. 2016 ; Harpaz et al. 2013 ; Kruspe et al. 2017 ). Signal detection is a critical step to increase sensitivity and resolution of methylation signal by reducing the signal-to-noise ratio and objectively controlling the false positive rate and prediction accuracy/risk. 6.1 Potential methylation signal The first estimation in our signal detection step is the identification of the cytosine sites carrying potential methylation signal P S . The methylation regulatory signal does not hold Weibull distribution and, consequently, for a given level of significance α (Type I error probability, e.g. α = 0.05), cytosine positions k with information divergence H k > = Hα =0.05 can be selected as sites carrying potential signals P S . The value of α can be specified. For example, potential signals with H k > Hα =0.01 can be selected. For each sample, cytosine sites are selected based on the corresponding fitted Weibull distribution model estimated in the previous step. Additionally, since cytosine with T V k = Hα =0.05 and T V k T V 0 , where T V 0 (‘tv.cut’) is a user specified value. The P S is detected with the function ‘getPotentialDIMP’: Download figure Open in new tab View this table: View inline View popup Notice that the total variation distance | T V | is an information divergence as well and it can be used in place of Hellinger divergence (Sanchez and Mackenzie 2016). The set of vectors P i = ( p i , 1 − p i ) and distance function | T V | integrate a metric space. In particular: That is, the quantitative effect of the vector components and (in our case, the effect of unmethylated read counts) is not present in T V as in . 6.2 Histogram and boxplots of methylation potential signals As before, a single GRanges object is built from the above set GRanges objects using the function ‘uniqueGRanges’, and the Hellinger divergences of the cytosine sites carrying P S (for each sample) are located in a single matrix on the metacolumn of the GRanges object. Download figure Open in new tab View this table: View inline View popup Download powerpoint Download figure Open in new tab Download figure Open in new tab Download figure Open in new tab 7 Cutpoint estimation Laws of statistical physics can account for background methylation, a response to thermal fluctuations that presumably functions in DNA stability (Sanchez and Mackenzie 2016). True signal is detected based on the optimal cutpoint ( López-Ratón et al. 2014 ), which can be estimated from the area under the curve (AUC) of a receiver operating characteristic (ROC) curve built from a logistic regression performed with the potential signals from controls and treatments. The ROC AUC is equivalent to the probability that a randomly-chosen positive instance is ranked more highly than a randomly-chosen negative instance ( Fawcett 2005 ). In the current context, the AUC is equivalent to the probability to distinguish a randomly-chosen methylation regulatory signal induced by the treatment from a randomly-chosen signal in the control. Download figure Open in new tab View this table: View inline View popup Download powerpoint Download figure Open in new tab View this table: View inline View popup Download powerpoint In practice, potential signals are classified as “control”’ ( CT ) and “treatment”’ ( T T ) signals (prior classification) and the logistic regression (LG): signal (with levels CT (0) and T T (1)) versus H k is performed. LG output yields a posterior classification for the signal. Prior and posterior classifications are used to build the ROC curve and then to estimate AUC and cutpoint H cutpoint . 8 DIMPs Cytosine sites carrying a methylation signal are designated differentially informative methylated positions (DIMPs). The probability that a DIMP is not induced by the treatment is given by the probability of false alarm ( P F A , false positive). That is, the biological signal is naturally present in the control as well as in the treatment. Each DIMP is a cytosine position carrying a significant methylation signal, which may or may not be represented within a differentially methylated position (DMP) according to Fisher’s exact test (or other current tests). A DIMP is a DNA cytosine position with high probability to be differentially methylated or unmethylated in the treatment with respect to a given control. Notice that the definition of DIMP is not deterministic in an ordinary sense, but stochastic-deterministic in physico-mathematical terms. DIMPs are selected with the function: Download figure Open in new tab 8.1 Histogram and boxplots of DIMPs The cutpoint detected with the signal detection step is very close (in this case) to the Hellinger divergence value H α=0.05 estimated for cancer tissue. The natural methylation regulatory signal is still present in a patient with cancer and reduced during the metastasis step. This signal is detected here as a false alarm ( P F A , false positive) The list of GRanges with DIMPs are integrated into a single GRanges object with the matrix of ‘hdiv’ values on its metacolumn: Download figure Open in new tab The multiplot with the histogram and the boxplot can now built: Download figure Open in new tab Download figure Open in new tab Download figure Open in new tab 8.2 Venn Diagram of DIMPs The Venn diagram of DIMPs reveals that the number cytosine site carrying methylation signal with a divergence level comparable to that observed in breast tissues with cancer and metastasis is relatively small (2797 DIMPs). The number of DIMPs decreased in the breast tissue with metastasis, but, as shown in the last boxplot, the intensity of the signal increased. Download figure Open in new tab Download figure Open in new tab Download figure Open in new tab Notice that natural methylation regulatory signals (not induced by the treatment) are present in both groups, control and treatment. The signal detection step permits us to discriminate the “ordinary” signals observed in the control from those induced by the treatment (a disease in the current case). In addition, this diagram reflects a classification of DIMPs only based on the cytosine positions. That is, this Venn diagram cannot tell us whether DIMPs at the same position can be distinguishable or not. For example, DIMPs at the same positions in control and treatment can happened with different probabilities estimated from their corresponding fitted Weibull distributions (see below). 8.3 Venn Diagram of DIMPs for reference Ref0 Download figure Open in new tab Download figure Open in new tab Download figure Open in new tab 9 Differentially informative methylated genomic regions (DIMRs) Our degree of confidence in whether DIMP counts in both groups of samples, control and treatment, represent true biological signal was determined in the signal detection step. To estimate DIMRs, we followed similar steps to those proposed in Bioconductor R package DESeq2 ( Love, Huber, and Anders 2014 ), but our GLM test looks for statistical difference between the groups based on gene-body DIMP counts overlapping a given genomic region rather than read counts. The regression analysis of the generalized linear model (GLM) with logarithmic link was applied to test the difference between group counts. The fitting algorithmic approaches provided by ‘glm’ and ‘glm.nb’ functions from the R packages stat and MASS, respectively, were used for Poisson (PR), Quasi-Poisson (QPR) and Negative Binomial (NBR) linear regression analyses, respectively. 9.1 Differentially methylated genes (DMGs) We shall call DMGs those DIMRs restricted to gene-body regions. DMGs are detected using function ‘countTest’. We used computational steps from DESeq2 packages. In the current case we follow the steps: Download figure Open in new tab Function ‘getDIMPatGenes’ is used to count the number of DIMPs at gene-body. The operation of this function is based on the ‘findOverlaps’ function from the ‘GenomicRanges’ Bioconductor R package. The ‘findOverlaps’ function has several critical parameters like, for example, ‘maxgap’, ‘minoverlap’, and ‘ignore.strand’. In our function ‘dimpAtGenes’, except for setting ignore.strand = TRUE and type = “within”, we preserve the rest of default ‘findOverlaps’ parameters. In this case, these are important parameter settings because the local mechanical effect of methylation changes on a DNA region where a gene is located is independent of the strand where the gene is encoded. That is, methylation changes located in any of the two DNA strands inside the gene-body region will affect the flexibility of the DNA molecule ( Choy et al. 2010 ; Severin et al. 2011 ). Download figure Open in new tab The number of DIMPs on the strand where a gene is encoded is obtained by setting ignore.strand = FALSE. However, for the current example results will be the same since the datasets downloaded from GEO do not have strand information. Next, the above GRanges objects carrying the DIMP counts from each sample are grouped into a single GRanges object. Since we have only one control, to perform group comparison and to move forward with this example, we duplicated ‘Breast_normal’ sample. Obviously, the confidence on the results increases with the number of sample replications per group (in this case, it is only an illustrative example on how to perform the analysis, since a fair comparison requires for more than one replicate in the control group). Download figure Open in new tab Next, the set of mapped genes are annotated Download figure Open in new tab Now, we build a ‘DESeqDataSet’ object using functions DESeq2 package. Download figure Open in new tab ## converting counts to integer mode DMG analysis is performed with the function ’countTest’ Download figure Open in new tab ## gene-wise dispersion estimates ## mean-dispersion relationship ## final dispersion estimates Download figure Open in new tab View this table: View inline View popup 9.2. DMGs for reference Ref0 Download figure Open in new tab ## converting counts to integer mode Download figure Open in new tab ## gene-wise dispersion estimates ## mean-dispersion relationship ## final dispersion estimates Download figure Open in new tab View this table: View inline View popup BRCA2, a breast cancer associated risk gene, is found between the DMGs Download figure Open in new tab View this table: View inline View popup Download powerpoint 10 Classification of DIMPs into two classes The regulatory methylation signal is an output from a natural process that continuously takes place across the ontogenetic development of the organism. Therefore, we expect to see methylation signal in natural, ordinary conditions. Function ‘evaluateDIMPclass’ can be used to perform a classification of DIMPs into two classes: DIMPS from control and DIMPs from treatment samples, as well as an evaluation of the classification performance (for more details see ?evaluateDIMPclass). In the setting below, a logistic regression: group versus divergence (at DIMPs), will be executed after randomly splitting the original DIMP dataset into two subsets: training (60%) and testing (40%). The performance of the logistic classifier using reference ‘Ref’ is: Download figure Open in new tab ## Model: treat ~ hdiv + TV + logP + pos Download figure Open in new tab View this table: View inline View popup Download powerpoint The best fitted logistic model using reference ‘Ref’ is: Download figure Open in new tab View this table: View inline View popup Download powerpoint In this case, the only variable not included in the model is total variation T V and all the rest are significant. The generalized linear regression can be performed by removing the variables T V . There are three other classifiers available: “pca.logistic”, “pca.lda”, and “pca.qda” (type ?evaluateDIMPclass in R console for more details). Principal component analysis (PCA) is used to convert a set of observations of possibly correlated predictor variables into a set of values of linearly uncorrelated variables (principal components, PCs). Then, the PCs are used as new uncorrelated predictor variables for LDA, QDA, and logistic classifiers. In the current case, the best classification result is obtained with the combination PCA + Quadratic Discriminant Analysis (PCA + QDA, “pca.qda”). Download figure Open in new tab ## Model: treat ~ hdiv + TV + logP + pos Download figure Open in new tab View this table: View inline View popup Download powerpoint Download figure Open in new tab View this table: View inline View popup Download powerpoint Monte Carlo (bootstrap) validation with 500 resamplings is performed by using the option ‘output = “mc.val”’: Download figure Open in new tab ## Model: treat ~ hdiv + TV + logP + pos conf.mat View this table: View inline View popup The performance of the PCA+QDA classifier using reference ‘Ref0’ is: Download figure Open in new tab ## Model: treat ~ hdiv + TV + logP + pos Download figure Open in new tab View this table: View inline View popup Download powerpoint Monte Carlo (bootstrap) validation with 500 resamplings using reference ‘Ref0’ can be now performed: Download figure Open in new tab ## Model: treat ~ hdiv + TV + logP + pos Download figure Open in new tab View this table: View inline View popup That is, with high accuracy level, DIMPs from control group can be discriminated from DIMPs found in cancer tissues. Supplements S1. Troubleshooting installation on Ubuntu Herein, a possible path to prevent potential issues originated during MethylIT installation on Ubuntu is given: To update R: To added an R CRAN repository typing in the terminal: sudo echo “deb /bin/linux/ubuntu xenial/” | sudo tee -a /etc/apt/sources.list For example: Download figure Open in new tab sudo apt update sudo apt upgrade Install Bioconductor: source(“ https://bioconductor.org/biocLite.R ”) biocLite() Install Bioconductor packages: ‘GenomicFeatures’, ‘VariantAnnotation’, ‘ensembldb’, ‘GenomicRanges’, ‘BiocParallel’, ‘biovizBase’, ‘DESeq2’, and ‘genefilter’. Package ‘GenomicFeatures’ depends on the R package ‘RMySQL’, which is not in ’Bioconductor. To install “RMySQL” from CRAN you might require the ’installation of the library “libmysqlclient-dev”. If this is the case, ’then you can solve it by typing in the Ubuntu Teminal: Download figure Open in new tab Next, in the R console: Download figure Open in new tab install.packages("devtools") devtools::install_git(" https://git.psu.edu/genomath/MethylIT ") S2. Session Information View this table: View inline View popup Acknowledgments We thank Professor David J Miller for valuable conversations and suggestions on our mathematical modeling. # Funding The work was supported by funding from NSF-SBIR (2015-33610-23428-UNL) and the Bill and Melinda Gates Foundation (OPP1088661). Footnotes rus547{at}psu.edu sam795{at}psu.edu References ↵ Baldi , Pierre , and Soren Brunak . 2001 . Bioinformatics: the machine learning approach . Second. Cambridge : MIT Press . ↵ Basu , A. , A. Mandal , and L. Pardo . 2010 . “ Hypothesis testing for two discrete populations based on the Hellinger distance .” Statistics & Probability Letters 80 ( 3–4 ). Elsevier B.V.: 206 – 14 . doi: 10.1016/j.spl.2009.10.008 . OpenUrl CrossRef ↵ Carter , Jane V. , Jianmin Pan , Shesh N. Rai , and Susan Galandiuk . 2016 . “ ROC-ing along: Evaluation and interpretation of receiver operating characteristic curves .” Surgery 159 ( 6 ). Mosby: 1638 – 45 . doi: 10.1016/j.surg.2015.12.029 . OpenUrl CrossRef PubMed ↵ Choy , Jahn S. , Sijie Wei , Ju Yeon Lee , Song Tan , Steven Chu , and Tae Hee Lee . 2010 . “ DNA methylation increases nucleosome compaction and rigidity .” Journal of the American Chemical Society 132 ( 6 ). American Chemical Society: 1782 – 3 . doi: 10.1021/ja910264z. OpenUrl CrossRef PubMed Web of Science ↵ Dolzhenko , Egor , and Andrew D Smith . 2014 . “ Using beta-binomial regression for high-precision differential methylation analysis in multifactor whole-genome bisulfite sequencing experiments .” BMC Bioinformatics 15 ( 1 ). BioMed Central: 215 . doi: 10.1186/1471-2105-15-215 . OpenUrl CrossRef PubMed ↵ Fawcett , Tom . 2005 . “ An introduction to ROC analysis .” doi: 10.1016/j.patrec.2005.10.010 . OpenUrl CrossRef Web of Science ↵ Greiner , M , D Pfeiffer , and R D Smith . 2000 . “ Principles and practical application of the receiveroperating characteristic analysis for diagnostic tests .” Preventive Veterinary Medicine 45 ( 1–2 ): 23 – 41 . doi: 10.1016/S0167-5877(00)00115-X . OpenUrl CrossRef PubMed Web of Science ↵ Harpaz , Rave , William DuMouchel , Paea LePendu , Anna Bauer-Mehren , Patrick Ryan , and Nigam H Shah . 2013 . “ Performance of Pharmacovigilance Signal Detection Algorithms for the FDA Adverse Event Reporting System .” Clin Pharmacol Ther 93 ( 6 ): 1 – 19 . doi: 10.1038/clpt.2013.24.Performance . OpenUrl CrossRef ↵ Hebestreit , Katja , Martin Dugas , and Hans-Ulrich Klein . 2013 . “ Detection of significantly differentially methylated regions in targeted bisulfite sequencing data .” Bioinformatics (Oxford, England) 29 ( 13 ): 1647 – 53 . doi: 10.1093/bioinformatics/btt263 . OpenUrl CrossRef PubMed Web of Science ↵ Kruspe , Sven , David D. Dickey , Kevin T. Urak , Giselle N. Blanco , Matthew J. Miller , Karen C. Clark , Elliot Burghardt , et al. 2017 . “ Rapid and Sensitive Detection of Breast Cancer Cells in Patient Blood with Nuclease-Activated Probe Technology .” Molecular Therapy Nucleic Acids 8 ( September ). Cell Press: 542 – 57 . doi: 10.1016/j.omtn.2017.08.004 . OpenUrl CrossRef ↵ Love , M I , W Huber , and S Anders . 2014 . “ Moderated estimation of fold change and dispersion for RNA-seq data with DESeq2 .” Genome Biology 15 ( 12 ): 1 – 34 . doi: Artn 550\rDoi 10.1186/S13059014-0550-8 . OpenUrl CrossRef PubMed ↵ López-Ratón , Mónica , Mar\’\ia Xosé Rodríguez-Álvarez , Carmen Cadarso-Suárez , Francisco GudeSampedro , and Others . 2014 . “ OptimalCutpoints: an R package for selecting optimal cutpoints in diagnostic tests .” Journal of Statistical Software 61 ( 8 ). Foundation for Open Access Statistics: 1 – 36 . https://www.jstatsoft.org/article/view/v061i08 . OpenUrl ↵ Robinson , Mark D. , Abdullah Kahraman , Charity W. Law , Helen Lindsay , Malgorzata Nowicka , Lukas M. Weber , and Xiaobei Zhou . 2014 . “ Statistical methods for detecting differentially methylated loci and regions .” Frontiers in Genetics 5 ( SEP ). Frontiers: 324 . doi: 10.3389/fgene.2014.00324 . OpenUrl CrossRef ↵ Barbara Bardoni Sanchez , Robersy , and Sally A. Mackenzie . 2016 . “ Information Thermodynamics of Cytosine DNA Methylation .” Edited by Barbara Bardoni . PLOS ONE 11 ( 3 ). Public Library of Science: e0150427 . doi: 10.1371/journal.pone.0150427 . OpenUrl CrossRef ↵ Sanchez , Robersy , Xiaodong Yang , Hardik Kundariya , Jose Raul Barreras , Yashitola Wamboldt , and Sally Mackenzie . 2018 . “ Enhancing resolution of natural methylome reprogramming behavior in plants .” bioRxiv , January . Cold Spring Harbor Laboratory, 252106 . doi: doi.org/10.1101/252106 . OpenUrl CrossRef ↵ Severin , Philip M D , Xueqing Zou , Hermann E Gaub , and Klaus Schulten . 2011 . “ Cytosine methylation alters DNA mechanical properties .” Nucleic Acids Research 39 ( 20 ): 8740 – 51 . doi: 10.1093/nar/gkr578 . OpenUrl CrossRef PubMed Web of Science ↵ Stevens , James P. 2009 . Applied Multivariate Statistics for the Social Sciences . Fifth Edit. Routledge Academic . Back to top Previous Next Posted April 03, 2018. Download PDF Email Thank you for your interest in spreading the word about bioRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following Example of Methylome Analysis with MethylIT using Cancer Datasets Message Subject (Your Name) has forwarded a page to you from bioRxiv Message Body (Your Name) thought you would like to see this page from the bioRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share Example of Methylome Analysis with MethylIT using Cancer Datasets Robersy Sanchez , Sally Mackenzie bioRxiv 261982; doi: https://doi.org/10.1101/261982 Share This Article: Copy Citation Tools Example of Methylome Analysis with MethylIT using Cancer Datasets Robersy Sanchez , Sally Mackenzie bioRxiv 261982; doi: https://doi.org/10.1101/261982 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Bioinformatics Subject Areas All Articles Animal Behavior and Cognition (8023) Biochemistry (18797) Bioengineering (14931) Bioinformatics (44508) Biophysics (22638) Cancer Biology (19780) Cell Biology (26957) Clinical Trials (138) Developmental Biology (14000) Ecology (21054) Epidemiology (2067) Evolutionary Biology (25492) Genetics (16188) Genomics (23537) Immunology (18737) Microbiology (42582) Molecular Biology (18110) Neuroscience (93654) Paleontology (701) Pathology (2989) Pharmacology and Toxicology (5105) Physiology (8133) Plant Biology (16030) Scientific Communication and Education (2098) Synthetic Biology (4574) Systems Biology (10256) Zoology (2393) window.__CF$cv$params={r:'a40258b29d742bef',t:'MTc5MDI1ODc1Mg==',u:'01a0d3bcc39b74ae89f41974dca4c898',ut:'hWqwJP877KSV3XoXzrSTzxUxAt2tg2jYlfVEwdQ6BYs-1790258758-1.2.1.1-ErrUtN5MK3C0On429tH5pnZACDPMJxklBJeutHd54tq.8YxNChMbxdBsH0fY5MSsUwiIhgLP.wYzqKnP6PVSfU1NJU_0Aj0bK6Bu26uhZkA',i:60};(function(){if(!document.body)return;var s=document.createElement('script');s.src='/cdn-cgi/challenge-platform/scripts/precursor/main.js';document.head.appendChild(s);})();

Text is read by the "Ask this paper" AI Q&A widget below. Extraction quality varies by source — PMC NXML preserves structure cleanly, OA-HTML may include some navigation residue, and OA-PDF can have broken hyphenation. The publisher copy (via DOI) is the canonical version.

My notes (saved in your browser only)

⚙ Ask this paper AI returns verbatim quotes from the full text · source: preprint-html ⓘ

Answers must be backed by verbatim quotes from this paper's full text. Hallucinated quotes are dropped automatically; if no verbatim passage answers the question, we say so. How this works

Citation neighborhood (no data yet)

We don't have any in-corpus citations linked to this paper yet. The paper's references may be in our DB but unresolved to ``paper_id`` (resolution happens at ingest when the cited DOI matches a row we already have). Run the cross-source citation reconcile pass to retry.

Source provenance

europepmc
last seen: 2026-05-19T01:45:01.086888+00:00
unpaywall
last seen: 2026-08-14T06:25:32.811723+00:00
License: CC-BY-NC-ND-4.0