Full text
24,738 characters
· extracted from
preprint-html
· click to expand
Damsel: Analysis and visualisation of DamID sequencing in R | bioRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-M677548'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search New Results Damsel: Analysis and visualisation of DamID sequencing in R View ORCID Profile Caitlin G Page , View ORCID Profile Andrew Londsdale , Katrina A Mitchell , Jan Schröder , View ORCID Profile Kieran F. Harvey , View ORCID Profile Alicia Oshlack doi: https://doi.org/10.1101/2024.06.12.598588 Caitlin G Page 1 Peter MacCallum Cancer Centre, Parkville, VIC , 3052, Australia 2 Sir Peter MacCallum Department of Oncology, The University of Melbourne, Parkville, VIC , 3052, Australia Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Caitlin G Page For correspondence: caitlin.page{at}petermac.org Andrew Londsdale 1 Peter MacCallum Cancer Centre, Parkville, VIC , 3052, Australia 2 Sir Peter MacCallum Department of Oncology, The University of Melbourne, Parkville, VIC , 3052, Australia 3 Murdoch Children’s Research Institute, Parkville, VIC , 3052, Australia Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Andrew Londsdale Katrina A Mitchell 1 Peter MacCallum Cancer Centre, Parkville, VIC , 3052, Australia 2 Sir Peter MacCallum Department of Oncology, The University of Melbourne, Parkville, VIC , 3052, Australia Find this author on Google Scholar Find this author on PubMed Search for this author on this site Jan Schröder 4 Computational Sciences Initiative, Department of Microbiology and Immunology, Peter Doherty Institute for Infection and Immunity, The University of Melbourne Find this author on Google Scholar Find this author on PubMed Search for this author on this site Kieran F. Harvey 1 Peter MacCallum Cancer Centre, Parkville, VIC , 3052, Australia 2 Sir Peter MacCallum Department of Oncology, The University of Melbourne, Parkville, VIC , 3052, Australia 5 Department of Anatomy and Developmental Biology, and Biomedicine Discovery Institute, Monash University Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Kieran F. Harvey Alicia Oshlack 1 Peter MacCallum Cancer Centre, Parkville, VIC , 3052, Australia 2 Sir Peter MacCallum Department of Oncology, The University of Melbourne, Parkville, VIC , 3052, Australia 6 School of Mathematics and Statistics, The University of Melbourne Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Alicia Oshlack Abstract Full Text Info/History Metrics Supplementary material Data/Code Preview PDF ABSTRACT Summary DamID sequencing is a technique to map the genome-wide interaction of a protein with DNA. Damsel is the first Bioconductor package to provide an end to end analysis for DamID sequencing data within R. Damsel performs quantification and testing of significant binding sites along with exploratory and visual analysis. Damsel produces results consistent with previous analysis approaches. Availability and implementation The R package Damsel is available for install through the Bioconductor project https://bioconductor.org/packages/release/bioc/html/Damsel.html and the code is available on GitHub https://github.com/Oshlack/Damsel/ Contact caitlin.page{at}petermac.org Supplementary information Available through the journal. INTRODUCTION Defining the binding landscape of chromatin interacting proteins is vital for understanding and manipulating gene expression. DamID was developed as an alternative to ChIP-seq that facilitates this without relying on antibodies ( van Steensel and Henikoff, 2000 ). In the DamID protocol, the E.coli DNA adenine methylase (Dam) protein is fused to a transcription regulatory protein of interest, and upon interaction with DNA, Dam methylates the adenine in adjacent GATC motifs. As Dam can also randomly methylate GATC motifs, DamID analysis is always conducted with the expression of a Dam-only control. Dam-methylated genomic regions are enriched using a methylation sensitive endonuclease to cut DNA and amplified with PCR. The ends of the resultant DNA fragments are sequenced to provide the genome-wide binding profile of the protein of interest. Following the identification of enriched genomic regions relative to the Dam-only control, potential target genes of the protein of interest can be identified. These target genes are identified based on their proximity to a peak. A limitation of DamID is the reliance on GATC motifs, which limits the utility of DamID to genetically tractable species. Targeted DamID is an optimised version of DamID that was developed in Drosophila melanogaster which allows in vivo expression and precise spatiotemporal control ( Southall et al., 2013 , Marshall et al., 2016 , Aughey et al., 2019 ). Despite DamID being a powerful alternative to ChIP-seq, there is limited availability of analytical tools. Bespoke analyses are commonly conducted, as in Vissers et al. (2018) which utilises edgeR’s ( Robinson et al., 2010 ) robust statistical testing. A 2019 review of DamID mentions only two published methods, both focussed on algorithms to facilitate peak calling (identifying regions of binding compared with the control) ( Aughey et al., 2019 ). Neither of these tools conduct the whole DamID bioinformatics workflow, thereby forcing researchers to rely on combining multiple tools and platforms with bespoke analysis. The widely used method, damidseq_pipeline ( Marshall and Brand, 2015 ), is the most comprehensive tool and runs on the command line analysing samples individually, but has no visualisation capabilities. Here, we present Damsel, the first dedicated R package for DamID, providing an end-to-end analysis, including exploratory and visualisation capabilities. METHOD The Damsel workflow, as described in Figure 1A , starts after read alignment using standard tools (Supplemental Methods) and requires BAM files and GATC regions for a genome as input. Damsel conducts the following steps in the analysis workflow: region quantification, identification of bound genomic regions, peak calling, candidate gene detection, and gene ontology testing, described in more detail below. Download figure Open in new tab Figure 1. Overview of Damsel. A The main steps in the Damsel workflow alongside their Bioconductor functions, including region quantification, identification of bound regions, peak calling, candidate gene detection, and gene ontology testing. B Example visualisation available in Damsel of the most significant peak found within the Vissers data. The layers of the plot show the counts across the regions for the Fusion and Dam-only samples, the differential methylation test results, the presence of significant peaks, the positions of the GATC sites, and the overlap to gene annotations. 1. Region quantification The genome is intrinsically partitioned into genomic regions demarcated by GATC motifs (note that GATC is its own reverse complement and therefore the motif demarcates both strands of the DNA simultaneously). Damsel then utilises the featureCount() function from the Rsubread package ( Liao et al., 2019 ) to summarise the read counts between two adjacent GATC sites (GATC region) from the provided BAM files and motif input file. For paired-end BAM files, Damsel instead summarises the fragments (read pairs). Due to the background methylation that occurs in the Dam-only sample, there is high correlation of the read counts of the regions between samples (Supplementary Fig. S1). 2. Identification of bound genomic regions Previously, enrichment of a bound genomic region was identified from the log2 ratio of the region counts of the fusion sample compared to the control. However, analysis by Marshall and Brand (2015) established that this can result in a negative bias, highlighting the importance of normalisation. Damsel first filters out excessively large regions (> 10 000 bp), along with low count regions, before normalisation and statistical testing is conducted with edgeR, incorporating the replicates for a streamlined analysis ( Robinson et al., 2010 ). edgeR’s TMM normalisation assumes the majority of regions are not significant ( Robinson and Oshlack 2010 ), and differential testing utilises a quasi-likelihood negative binomial model ( Robinson et al., 2010 ), utilising experimental replicates to model the biological variance in the data. 3. Peak calling Peak calling allows for aggregation of adjacent significant regions that are likely to be the result of the same binding event. Damsel identifies peaks by aggregating the region level differential expression results, with p-value threshold set to 0.01 (by default), and log2 fold change threshold set to one. If the gap to the next significant region is less than 150bp, it is included in the peak. Regions smaller than this size have consistently lower counts, and thus represent a greater proportion of regions that are excluded from differential testing. Peaks are ranked based on the statistics for ChIP-seq analysis presented in csaw ( Lun and Smyth 2016 ), using the region with the lowest p-value to represent the peak and ordering peak significance based on this p-value. 4. Candidate gene identification Methods for associating peaks and genes match the peak to its closest gene, with distance taken from the middle of the peak to the transcription start site. This is straightforward if the analysis is conducted on a species with little overlap between genes, like humans. However in species like Drosophila melanogaster , there is a large amount of overlap between genes and the closest gene may not be the most likely candidate. Therefore, Damsel outputs each peak with a list of candidate target genes, along with their genomic positions and relevant statistics. Placing the genes within the context of the peaks provides a more informative output, and sets Damsel apart from other tools. At this point, Damsel outputs a file with all identified peaks and their associated statistics, matched to their associated candidate genes. Comparatively, damidseq_pipeline ( Marshall and Brand 2015 ) outputs a list of genes and a score, which is the end of the analysis. However, Damsel also includes the following utilities. 5. Gene ontology testing Gene ontology testing can be used to interpret the set of candidate target genes identified from the peaks. Damsel repurposes goseq, an R package developed to correct for detection biases in RNA-seq ( Young et al., 2010 ). This is important because the GATC motifs are not spread evenly throughout the genome and different genes have different lengths and therefore a different number of GATC regions associated with them. This results in a bias where genes with more GATC regions are more likely to be associated with a peak (binding event) (Supplementary Fig. S2). Without bias correction, the gene ontology terms identified are not truly representative of the data. goseq takes as input a list of genes, if they are significant, and the number of GATC regions within 2kb of the gene as the bias parameter, and outputs a list of overrepresented gene ontology terms. 7. Visualisations A unique feature of Damsel is its visualisation capability. Other DamID methods, including damidseq_pipeline ( Marshall and Brand 2015 ), output files that they recommend viewing in IGV, requiring researchers to switch software platforms. Damsel provides the opportunity to use its output files to create plots within R. Building on ggbio ( Yin et al., 2012 ) and ggcoverage ( Song and Yang 2023 ), Damsel contains ggplot2 ( Wickham 2010 ) style plots that present results from different stages of the analysis. These plots can be layered, allowing the visualisation of counts per region, log2 fold change and differential expression testing p-values, peak location, GATC motif, and gene positions for a provided region ( Figure 1.B ). RESULTS The existing approaches for DamID analysis ( Marshall and Brand (2015) , Vissers et al., (2018)) conduct a similar sequence of stages as Damsel including: identifying regions of enrichment, combining regions into peaks, and associating peaks with genes (Supplementary Table 1). Some of these stages are handled differently in Damsel with increased functionality and seamless data flow through the pipeline. Like Vissers’ (2018) analysis, Damsel incorporates replicates into the analysis to model variability and significantly reduce the burden of compiling results. Additional features unique to Damsel include region quantification directly from bam files, the ranking of significant results (peaks and genes), providing relevant statistics associating the peaks with annotated genes, and capacity for visualisation of the results (Supplementary Table 1). To evaluate Damsel, we applied it to a subset of the data presented in Vissers et. al. (2018). To evaluate the false positive rate, the samples from this experiment were randomly assigned as Fusion and Dam-only. When a Dam-only and Fusion sample from different replicates were assigned as Dam-only (or Fusion), the results were as expected - with no regions identified as significant at FDR < 0.01, and therefore no peaks were identified. This demonstrates good false discovery rate control. Similarly the Vissers’ pipeline and the Marshall and Brand pipeline (2015) found no significant peaks in this scenario. We next analysed the data in the way it was intended with two replicates of the fusion and control. Damsel was found to be more sensitive than the original reported results finding 2,955 peaks, which is 669 more than identified from running Vissers’ (2018) pipeline. Unsurprisingly, Damsels results were more similar to the Visser pipeline than to DamID-seq pipeline with more than 60% of peaks overlapping with Visser (Supplementary Figure S3). We performed Damsel’s GO testing on the peaks reported in the original paper and those found by Damsel and reassuringly found that Damsel reported 9 of the same top 10 GO ontology terms as the regions from Vissers (Supplementary Figure S4). Finally, as an example of data visualisation we used Damsel to show the data and analysis of the most significant peak identified by all three methods ( Figure 1B ). Identifying which method is the most sensitive and accurate is beyond the scope of this study, and would be difficult without a large amount of experimental validation. However, Damsel performs consistently compared with established methods, and has produced results on multiple published datasets. CONCLUSION Damsel offers an end-to-end DamID analysis pipeline including exploratory analysis and visualisations. On published data we find similar results to previous bespoke analysis. We find that Damsel can control false discovery rate and is more sensitive. Damsel includes new features such as the ranking of peak significance and performing gene ontology analysis. Future work will include the utility to perform motif enrichment analysis on sequences found in enriched peaks. Data availability The data used in this paper is available at the European Nucleotide Archive (PRJNA494322) https://www.ebi.ac.uk/ena/browser/view/PRJNA494322 Funding A.O was supported by an Investigator grant (APP1196256) from the National Health and Medical Research Council of Australia. K.F.H was supported by an Investigator grant (APP1194467) from the National Health and Medical Research Council of Australia. Conflict of Interest: none declared Acknowledgements We thank Avni Avand for testing the software and providing feedback. Footnotes https://www.ebi.ac.uk/ena/browser/view/PRJNA494322 References ↵ Aughey GN , Cheetham SW , Southall TD . DamID as a versatile tool for understanding gene regulation . Development 2019 ; 146 ( 6 ). Chen Y , Lun ATL , Smyth GK ( 2016 ). From reads to genes to pathways: differential expression analysis of RNA-Seq experiments using Rsubread and the edgeR quasi-likelihood pipeline . F1000Research, 2016 ; 5 , 1438 . doi: 10.12688/f1000research.8987.2 . OpenUrl CrossRef PubMed Chen Y et al. edgeR 4.0: powerful differential analysis of sequencing data with expanded functionality and improved support for small counts and larger datasets . bioRxiv 2024 . doi: 10.1101/2024.01.21.576131 . OpenUrl Abstract / FREE Full Text ↵ Liao Y , Smyth GK , Shi W. The R package Rsubread is easier, faster, cheaper and better for alignment and quantification of RNA sequencing reads . Nucleic Acids Research , 2019 ; 47 , e47 . doi: 10.1093/nar/gkz114 . OpenUrl CrossRef PubMed Lun ATL , Smyth GK . De novo detection of differentially bound regions for ChIP-seq data using peaks and windows: controlling error rates correctly . Nucleic Acids Res ., 2014 ; 42 ( 11 ), e95 . OpenUrl CrossRef PubMed ↵ Lun ATL , Smyth GK . csaw: a Bioconductor package for differential binding analysis of ChIP-seq data using sliding windows . Nucleic Acids Res 2016 ; 44 ( 5 ), e45 . OpenUrl CrossRef PubMed ↵ Marshall OJ , Brand AH . damidseq_pipeline: an automated pipeline for processing DamID sequencing datasets . Bioinformatics 2015 ; 31 ( 20 ): 3371 – 3373 OpenUrl CrossRef PubMed ↵ Marshall OJ et al. Cell-type-specific profiling of protein–DNA interactions without cell isolation using targeted DamID with next-generation sequencing . Nature protocols 2016 ; 11 : 1586 – 1598 OpenUrl McCarthy DJ , Chen Y , Smyth GK ( 2012 ). Differential expression analysis of multifactor RNA-Seq experiments with respect to biological variation . Nucleic Acids Research, 2012 ; 40 ( 10 ): 4288 – 4297 . doi: 10.1093/nar/gks042 . OpenUrl CrossRef PubMed Web of Science ↵ Robinson MD , McCarthy DJ , Smyth GK . edgeR: a Bioconductor package for differential expression analysis of digital gene expression data . Bioinformatics , 2010 ; 26 ( 1 ), 139 – 140 . doi: 10.1093/bioinformatics/btp616 . OpenUrl CrossRef PubMed Web of Science ↵ Robinson MD , Oshlack A. A scaling normalization method for differential expression analysis of RNA-seq data . Genome Biol 11 , 2010 ; R25 doi: 10.1186/gb-2010-11-3-r25 OpenUrl CrossRef PubMed ↵ Song Y , Wang J. ggcoverage: an R package to visualize and annotate genome coverage for various NGS data . BMC Bioinformatics 2023 ; 24 : 309 . doi: 10.1186/s12859-023-05438-2 OpenUrl CrossRef ↵ Southall TD et al. Cell-Type-Specific Profiling of Gene Expression and Chromatin Binding without Cell Isolation: Assaying RNA Pol II Occupancy in Neural Stem Cells . Developmental Cell 2013 ; 26 ( 1 ): P101 – 112 OpenUrl ↵ van Steensel B , Henikoff S. Identification of in vivo DNA targets of chromatin proteins using tethered Dam methyltransferase . Nature Biotechnology 2000 ; 18 : 424 – 428 OpenUrl CrossRef PubMed Web of Science ↵ Vissers JHA et al. The Scalloped and Nerfin-1 Transcription Factors Cooperate to Maintain Neuronal Cell Fate . Cell Reports . 2018 ; 25 ( 6 ): 1561 – 1576 OpenUrl ↵ Wickham H ggplot2: Elegant graphics for data analysis . Springer Cham , 2016 ↵ Yin T , Cook D , Lawrence M. “ ggbio: an R package for extending the grammar of graphics for genomic data .” Genome Biology , 2012 ; 13 ( 8 ), R77 . OpenUrl CrossRef PubMed ↵ Young MD , Wakefield MJ , Smyth GK , Oshlack A. Gene ontology analysis for RNA-seq: accounting for selection bias . Genome Biology , 2010 ; 11 , R14 . OpenUrl CrossRef PubMed View the discussion thread. Back to top Previous Next Posted June 14, 2024. Download PDF Supplementary Material Data/Code Email Thank you for your interest in spreading the word about bioRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following Damsel: Analysis and visualisation of DamID sequencing in R Message Subject (Your Name) has forwarded a page to you from bioRxiv Message Body (Your Name) thought you would like to see this page from the bioRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share Damsel: Analysis and visualisation of DamID sequencing in R Caitlin G Page , Andrew Londsdale , Katrina A Mitchell , Jan Schröder , Kieran F. Harvey , Alicia Oshlack bioRxiv 2024.06.12.598588; doi: https://doi.org/10.1101/2024.06.12.598588 Share This Article: Copy Citation Tools Damsel: Analysis and visualisation of DamID sequencing in R Caitlin G Page , Andrew Londsdale , Katrina A Mitchell , Jan Schröder , Kieran F. Harvey , Alicia Oshlack bioRxiv 2024.06.12.598588; doi: https://doi.org/10.1101/2024.06.12.598588 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Bioinformatics Subject Areas All Articles Animal Behavior and Cognition (7644) Biochemistry (17728) Bioengineering (13917) Bioinformatics (42038) Biophysics (21489) Cancer Biology (18637) Cell Biology (25553) Clinical Trials (138) Developmental Biology (13401) Ecology (19941) Epidemiology (2067) Evolutionary Biology (24367) Genetics (15622) Genomics (22547) Immunology (17764) Microbiology (40475) Molecular Biology (17208) Neuroscience (88749) Paleontology (667) Pathology (2842) Pharmacology and Toxicology (4834) Physiology (7659) Plant Biology (15175) Scientific Communication and Education (2047) Synthetic Biology (4304) Systems Biology (9835) Zoology (2272)
Text is read by the "Ask this paper" AI Q&A widget below.
Extraction quality varies by source — PMC NXML preserves structure
cleanly, OA-HTML may include some navigation residue, and OA-PDF can
have broken hyphenation. The publisher copy
(via DOI)
is the canonical version.