The significance of molecular heterogeneity in breast cancer batch correction and dataset integration

preprint OA: closed

Abstract

Breast cancer research benefits from a substantial collection of gene expression datasets that are commonly integrated to increase analytical power. Gene expression batch effects arising between experimental batches, where signal differences confound true biological variation, must be addressed when integrating datasets and several approaches exist to address these technical differences. This brief communication study clearly demonstrates that popular batch correction techniques can significantly distort key biomarker expression signals. Through the implementation of ComBat batch correction and evaluation of integrated expression values, we profile the extent of these distortions and consider an additional mitigatory batch correction step. We demonstrate that leveraging a priori knowledge of sample molecular subtype classification can optimally remove batch effect distortion while preserving key biomarker expression variation and transcriptional legitimacy. To the best of our knowledge, this study presents the first analysis of the interplay between dataset molecular composition and the concomitant robustness of integrated, batch-corrected biological expression signal.
Full text 30,397 characters Β· extracted from preprint-html Β· click to expand
The significance of molecular heterogeneity in breast cancer batch correction and dataset integration | medRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-P4HH5NV'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search The significance of molecular heterogeneity in breast cancer batch correction and dataset integration View ORCID Profile Nicholas Moir , Dominic A. Pearce , Simon P. Langdon , View ORCID Profile T. Ian Simpson doi: https://doi.org/10.1101/2024.12.22.24319524 Nicholas Moir 1 Applied Bioinformatics of Cancer, University of Edinburgh Cancer Research Centre, Institute of Genetics and Cancer , Edinburgh, UK 2 Research Services, Information Services, University of Edinburgh , UK Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Nicholas Moir For correspondence: nicholas.moir{at}ed.ac.uk Dominic A. Pearce 3 Fios Genomics Bioquarter , 13 Little France Rd., Edinburgh UK Find this author on Google Scholar Find this author on PubMed Search for this author on this site Simon P. Langdon 4 Edinburgh Cancer Research and Edinburgh Pathology, Institute of Genetics and Cancer, University of Edinburgh , UK Find this author on Google Scholar Find this author on PubMed Search for this author on this site T. Ian Simpson 5 School of Informatics, University of Edinburgh , Edinburgh, UK Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for T. Ian Simpson Abstract Full Text Info/History Metrics Data/Code Preview PDF Abstract Breast cancer research benefits from a substantial collection of gene expression datasets that are commonly integrated to increase analytical power. Gene expression batch effects arising between experimental batches, where signal differences confound true biological variation, must be addressed when integrating datasets and several approaches exist to address these technical differences. This brief communication study clearly demonstrates that popular batch correction techniques can significantly distort key biomarker expression signals. Through the implementation of ComBat batch correction and evaluation of integrated expression values, we profile the extent of these distortions and consider an additional mitigatory batch correction step. We demonstrate that leveraging a priori knowledge of sample molecular subtype classification can optimally remove batch effect distortion while preserving key biomarker expression variation and transcriptional legitimacy. To the best of our knowledge, this study presents the first analysis of the interplay between dataset molecular composition and the concomitant robustness of integrated, batch-corrected biological expression signal. Main text The reliability and robustness of oncology gene expression profiling studies are affected by the size and quality of available sample data. Sample count has a major impact on the reliability of clinical gene expression analyses, yet the size of most previous studies has been driven by sample availability and cost. Nonetheless, enhanced biological insight can be empowered by integrating discrete datasets into a larger meta-dataset. The improved statistical power of downstream integrated analysis has been demonstrated 1 –6 and leveraged to identify and categorise molecular events within both breast cancer 7 and the wider oncological research landscape 7 –10 that may be miscategorised or undetectable in smaller datasets. Direct integration of probe or transcript-level expression data from multiple studies is therefore potentially very powerful but technical batch-effects, manifesting both within and between studies, must be addressed before initiating integrated analysis 1 –3,11–14. Several groups have investigated optimal batch correction approaches, where expression data are augmented to remove batch effects while ensuring the maximal retention of signals representative of true biological variation. To this end, ourselves and others have previously stated that breast cancer datasets should only be integrated where they are suitably β€˜similar’ 2 , but enhanced understanding of these fundamental issues still appear elusive many years later. Additionally, we have cautioned that the molecular composition of breast cancer expression data does not accurately reflect breast cancer heterogeneity at the population level 15 , and have previously highlighted that integrating datasets with vastly varying molecular compositions can dramatically reduce the accuracy of prognosis predictions 2 . Batch-effect correction has gained acceptance as a necessary dataset integration step but there remains little focus on the potential interplay between sample molecular subtype composition and batch correction robustness. In this study, we explore the effect of ComBat 16 batch correction on fidelity of key breast cancer biomarker expression signals, and show that consideration of molecular heterogeneity is required to ensure the optimal preservation of vital gene expression signal integrity. METABRIC data were collected between 1977 and 2005 from five centres in the UK and Canada and packaged as two stand-alone discovery ( n=997 ) and validation (n=995) 17 annotated expression datasets. The potential integration of these into a single 1992 sample dataset is of obvious analytical benefit in downstream molecular subtype and novel biomarker discovery pipeline workflows 18 , 19 . However, the existence of strong batch effects between the discovery and validation datasets has been cautioned against in published studies 18 , 20 , 21 . As described, effective batch correction removes technical differences without undesirable further modification of the biological signal. In this brief communication, we evaluate the performance of several different ComBat implementations on expression values of the ILMN_1678535 (ESR1) 22 , ILMN_2352131 (ERBB2) 23 and ILMN_1680955 (AURKA) 24 probes. These genes comprise the SCMGENE subtype classification model 25 and are recognised as key breast cancer biomarkers of hormonal signalling and proliferation 25 , 26 . To investigate and illustrate the augmentation of these expression values during batch correction, we compare biomarker expression within the canonical PAM50 molecular subtypes (Basal, Luminal A, Luminal B, Her2+ and Normal Breast-like) between METABRIC discovery and validation datasets. Significant expression differences are expected between uncorrected data given the acknowledged presence of batch effects and, recognising sample variation as a sum of technical and biological differences, insight into the relative effectiveness of each ComBat approach is gained by comparing post-correction expression within each molecular subtype class. Substantial inter-sample molecular differences are encapsulated within PAM50 subtype assignment, and minimisation of significant variation within subtype classes therefore acts as an appropriate proxy for assessing and comparing batch correction efficacy. In this study, we compare the effect of four different ComBat implementations on biomarker expression. ComBat implements a location-scale (LS) method, which models expression mean and variance within each batch, before adjusting expression according to these models. Default parametric correction assumes that the location batch effect variables originate from the same normal distribution and scale effect variables from the same inverse gamma distribution. Non-parametric correction, which uses a Monte Carlo integration-based technique to estimate the location and scale batch effect parameters 27 is also evaluated. In addition to these parametric and non-parametric corrections, we include two approaches that account for dataset molecular heterogeneity by leveraging a priori PAM50 subtype assignment. We investigate including PAM50 molecular subtype as a defined covariate in the model used by ComBat and additionally introduce a further approach that leverages a pre-correction subtype stratification of samples. Subtype-stratified batch correction involves pooling samples of each molecular subtype from each batch and correcting each PAM50 molecular subtype group separately. Transcriptome biomarker fidelity, represented as a minimisation of within-subtype sample expression difference following correction, is clearly enhanced when sample molecular heterogeneity is considered during batch correction. Statistically significant (two-sided Mann-Whitney-Wilcoxon test followed by Holm-Bonferroni correction) Aurora Kinase A (AURKA) expression ( figure 1A ) differences persists in both Luminal A (parametric p=.001, non-parametric p=.001) and Luminal B (parametric p<.001, non-parametric p<.001) samples following both parametric and non-parametric ComBat correction. In contrast, both a priori PAM50 batch correction approaches remove technical differences and result in no significant population expression differences between discovery and validation sample groups. The benefit of leveraging molecular subtype assignment is further demonstrated with Oestrogen receptor alpha (ESR1) expression ( figure 1B ). Rather disturbingly, uncorrected basal samples display no significant population difference but have significant population expression differences introduced through parametric (p<.001), non-parametric (p<.001) and PAM50 covariate correction (p<.001). HER2 samples retain differences following all four parametric (p<.001), non-parametric (p<.001), PAM50 stratified (p=.019) and PAM50 covariate (p<.001) correction approaches. Within luminal A samples, parametric (p<.001) and non-parametric (p<.001) perform poorly, while both a priori PAM50 approaches remove significant ESR1 expression differences between sample populations. Similarly, both parametric (p<.001) and non-parametric (p<.001) fail to resolve population differences in Luminal B samples, with PAM50 stratified correction retaining statistical significance between batches (p=.019) and only PAM50 covariate correction removing statistical significance. Parametric (p<.001) and non-parametric (p<.001) correction additionally performs poorly with normal-like samples, with both PAM50-based approaches removing significant population differences, a prerequisite for accurate post-integration analysis. Download figure Open in new tab Figure 1. Effect of various ComBat batch correction implementations on METABRIC biomarker expression. Comparison of sample expression values before and after 5 batch corrections in AURKA ( A ), ESR1 ( B ) and ERBB2 ( C) . Expression values are compared between METABRIC discovery and validation batches within PAM50 molecular subtype classifications, allowing visualisation of potential batch correction signal augmentation. Standard non-parametric and parametric correction appear worse at minimising between-batch differences within each PAM50 molecular subtype. The addition of PAM50 molecular subtype as a covariate can improve correction while a stratified PAM50 approach removes batch effects most effectively. Pairwise comparisons between batches were performed with a two-sided Mann-Whitney-Wilcoxon test followed by Holm-Bonferroni correction. Non-significant (p > .05) comparisons are unlabelled while * denotes p <= .05, ** denotes p <= .01, *** denotes p <= .001 and **** denotes p <= .0001. The third and final biomarker evaluated is expression of erbb2 receptor tyrosine kinase 2 (ERBB2). Significant expression difference between batches are introduced to HER2 PAM50 samples by parametric (p=.011) and non-parametric (p=.014) correction. Luminal A samples have between-batch significance removed by all correction techniques except PAM50 covariate correction (p=.046). Leveraging the METABRIC discovery and validation datasets, we have outlined the importance of considering molecular subtype when batch-correcting before dataset integration. An additional interesting scenario arises when an individual dataset is produced at various time-points or by multiple laboratories, introducing potential batch effect manifestation. To investigate this premise, we consider a within-dataset evaluation of GSE6532 28 , a published dataset used in multiple studies 29 –31 that has been identified as displaying technical batch effects 32 . Samples within GSE6532 were ComBat corrected using the four approaches previously described. Following batch correction, augmentation of GSE6532 ESR1 expression ( figure 2 ) displays a similar outcome to METABRIC correction, with fewer significant differences following PAM50 covariate and stratified approaches. Basal samples, located only in batches one and two within this dataset, have significant population expression differences introduced following parametric (p=.006) and non-parametric (p=.022) correction. Luminal B samples display significant population differences following parametric correction (batches 2&3, p=.024 and batches 1&3, p=.024), non-parametric (between batches 2&3, p=.011 and 1&3, p=.011) and PAM50 covariate ComBat (batches 2&3, p=.05), with only PAM50 stratified correction displaying no significance between corrected batches. Normal-like samples continue to display significant differences following parametric (batches 1&3, p<.001 and batches 1&3, p=.014) and non-parametric (batches 1&3, p<.001 and batches 2&3, p=.023) ComBat. Download figure Open in new tab Figure 2. Effect of various ComBat batch correction implementations on GSE6532 ESR1 biomarker expression. Comparison of sample ESR1 expression values before and after 5 batch corrections. Samples were identified as belonging to three batches based on original sample processing date and stratified according to PAM50 molecular subtype class. Expression values are compared between batches within each subtype class following batch correction to visualise correction effects. ComBat non-parametric and parametric correction introduces significant signal difference between batches in Basal samples and performs poorly in Luminal B and Normal samples. Recognition of molecular subtype improves performance, with PAM50 stratified correction removing batch effects most effectively. Pairwise comparisons between patient groups were performed with a two-sided Mann-Whitney-Wilcoxon test followed by Holm-Bonferroni correction.Non-significant (p > .05) comparisons are unlabelled while * denotes p <= .05, ** denotes p <= .01, *** denotes p <= .001 and **** denotes p <= .0001. The data presented herein illustrate often underappreciated subtleties in batch effect correction. Despite being a vital prerequisite for powerful data integration analysis, undesirable augmentation of molecular signal has the potential to seriously impact integrated analysis. Departures in molecular expression measurements between batches can be reasoned as composites of technical batch effects and biological, demonstrated by sample molecular subtype composition, structured differences. While the goal of batch correction is to address the former while retaining the latter, and acknowledging that no method will ever perfectly dissect these components, within this brief communication we highlight that additional consideration must be given to potential sub-optimal expression differences persisting following correction. The popular ComBat algorithm typically assumes identical distribution of gene expression across batches, but this assumption can be violated by divergent sample transcriptome profiles characteristic of molecularly heterogeneous cancers such as breast. To investigate this effect, we compared both standard parametric and non-parametric ComBat with PAM50-leveraged correction approaches. Our results show that encapsulation of molecular heterogeneity can help optimise the desired removal of technical effects while preserving true biological signal. It is perhaps reasonable and logical that incorporating molecular subtype classes within batch correction pipelines provide an effective shielding of molecular heterogeneity but, to our knowledge, this has not been elucidated until now. Within this article we purposefully refrain from definitively suggesting a particular batch correction approach. Rather, we invite consideration of an often overlooked, but potentially serious, feature of molecularly heterogeneous gene expression analysis that has the potential to seriously impact analysis and resulting scientific inference 33 . Data availability METABRIC datasets are available via committee approval. GSE6532 is freely available at https://www.ncbi.nlm.nih.gov/geo . Conflicts of Interest The authors declare no conflicts of interest. Additional Information Our colleague Dr. Andrew H. Sims sadly passed away during the course of this project. He was instrumental in both the conceptualisation and supervision of this study. References 1. ↡ Leek , J. T. et al. Tackling the widespread and critical impact of batch effects in high-throughput data . Nat Rev Genet 11 , 733 – 739 ( 2010 ). OpenUrl CrossRef PubMed Web of Science 2. ↡ Sims , A. H. et al. The removal of multiplicative, systematic bias allows integration of breast cancer gene expression datasets - improving meta-analysis and prediction of prognosis . BMC Med Genomics 1 , 42 ( 2008 ). OpenUrl CrossRef PubMed 3. Lazar , C. et al. Batch effect removal methods for microarray gene expression data integration: a survey . Brief Bioinform 14 , 469 – 490 ( 2013 ). OpenUrl CrossRef PubMed 4. Turnbull , A. K. et al. Direct integration of intensity-level data from Affymetrix and Illumina microarrays improves statistical power for robust reanalysis . BMC Med Genomics 5 , 35 ( 2012 ). OpenUrl CrossRef PubMed 5. Shabalin , A. A. , Tjelmeland , H. , Fan , C. , Perou , C. M. & Nobel , A. B. Merging two gene-expression studies via cross-platform normalization . Bioinformatics 24 , 1154 – 1160 ( 2008 ). OpenUrl CrossRef PubMed Web of Science 6. Benito , M. et al. Adjustment of systematic microarray data biases . Bioinformatics 20 , 105 – 114 ( 2004 ). OpenUrl CrossRef PubMed Web of Science 7. ↡ Zhang , D. , Wang , Y. , Zhao , F. & Yang , Q. Integrated multiomics analyses unveil the implication of a costimulatory molecule score on tumor aggressiveness and immune evasion in breast cancer: A large-scale study through over 8,000 patients . Computers in Biology and Medicine 159 , 106866 ( 2023 ). OpenUrl PubMed 8. He , J. et al. Elevated expression of glycolytic genes as a prominent feature of early-onset preeclampsia: insights from integrative transcriptomic analysis . Frontiers in Molecular Biosciences 10 , ( 2023 ). 9. Rokavec , M. , Γ–zcan , E. , Neumann , J. & Hermeking , H. Development and Validation of a 15-gene Expression Signature with Superior Prognostic Ability in Stage II Colorectal Cancer . Cancer Research Communications 3 , 1689 – 1700 ( 2023 ). OpenUrl PubMed 10. Yu , X. , Liu , Y. & Chen , M. Reassessment of Reliability and Reproducibility for Triple-Negative Breast Cancer Subtyping . Cancers 14 , 2571 ( 2022 ). OpenUrl PubMed 11. Parker , H. S. & Leek , J. T. The practical effect of batch on genomic prediction . Statistical Applications in Genetics and Molecular Biology 11 , ( 2012 ). 12. Kitchen , R. R. et al. Correcting for intra-experiment variation in Illumina BeadChip data is necessary to generate robust gene-expression profiles . BMC Genomics 11 , 134 ( 2010 ). OpenUrl CrossRef PubMed 13. Chen , C. et al. Removing Batch Effects in Analysis of Expression Microarray Data: An Evaluation of Six Batch Adjustment Methods . PLoS ONE 6 , ( 2011 ). 14. Tseng , G. C. , Ghosh , D. & Feingold , E. Comprehensive literature review and statistical considerations for microarray meta-analysis . Nucleic Acids Res 40 , 3785 – 3799 ( 2012 ). OpenUrl CrossRef PubMed Web of Science 15. ↡ Xie , Y. et al. Breast cancer gene expression datasets do not reflect the disease at the population level . npj Breast Cancer 6 , 1 – 5 ( 2020 ). OpenUrl CrossRef PubMed 16. ↡ Leek , J. T. , Johnson , W. E. , Parker , H. S. , Jaffe , A. E. & Storey , J. D. The sva package for removing batch effects and other unwanted variation in high-throughput experiments . Bioinformatics 28 , 882 – 883 ( 2012 ). OpenUrl CrossRef PubMed Web of Science 17. ↡ Pereira , B. et al. The somatic mutation profiles of 2,433 breast cancers refine their genomic and transcriptomic landscapes . Nat Commun 7 , 11479 ( 2016 ). OpenUrl CrossRef PubMed 18. ↡ Vidula , N. , Yau , C. , Wolf , D. & Rugo , H. S. Androgen receptor gene expression in primary breast cancer . NPJ Breast Cancer 5 , 47 ( 2019 ). OpenUrl PubMed 19. ↡ Paquet , E. R. , Lesurf , R. , Tofigh , A. , Dumeaux , V. & Hallett , M. T. Detecting gene signature activation in breast cancer in an absolute, single-patient manner . Breast Cancer Research 19 , 32 ( 2017 ). OpenUrl PubMed 20. ↡ Mucaki , E. J. et al. Predicting Outcomes of Hormone and Chemotherapy in the Molecular Taxonomy of Breast Cancer International Consortium (METABRIC) Study by Biochemically-inspired Machine Learning . F1000Res 5 , 2124 ( 2017 ). OpenUrl 21. ↡ Dutta , D. , Sen , A. & Satagopan , J. Sparse canonical correlation to identify breast cancer related genes regulated by copy number aberrations . PLOS ONE 17 , e0276886 ( 2022 ). OpenUrl PubMed 22. ↡ ESR1 - Estrogen receptor - Identifiers . https://www.nextprot.org/entry/NX_P03372/identifiers . 23. ↡ ERBB2 - Receptor tyrosine-protein kinase erbB-2 - Identifiers . https://www.nextprot.org/entry/NX_P04626/identifiers . 24. ↡ AURKA - Aurora kinase A - Identifiers . https://www.nextprot.org/entry/NX_O14965/identifiers . 25. ↡ Haibe-Kains , B. et al. A Three-Gene Model to Robustly Identify Breast Cancer Molecular Subtypes . JNCI: Journal of the National Cancer Institute 104 , 311 – 325 ( 2012 ). OpenUrl CrossRef PubMed Web of Science 26. ↡ Desmedt , C. et al. Biological Processes Associated with Breast Cancer Clinical Outcome Depend on the Molecular Subtypes . Clinical Cancer Research 14 , 5158 – 5165 ( 2008 ). OpenUrl Abstract / FREE Full Text 27. ↡ Johnson , W. E. , Li , C. & Rabinovic , A. Adjusting batch effects in microarray expression data using empirical Bayes methods . Biostatistics 8 , 118 – 127 ( 2007 ). OpenUrl CrossRef PubMed Web of Science 28. ↡ GEO Accession viewer . https://www.ncbi.nlm.nih.gov/geo/query/acc.cgi?acc=GSE6532 . 29. ↡ Loi , S. et al. Gene expression profiling identifies activated growth factor signaling in poor prognosis (Luminal-B) estrogen receptor positive breast cancer . BMC Med Genomics 2 , 37 ( 2009 ). OpenUrl CrossRef PubMed 30. Loi , S. et al. Definition of clinically distinct molecular subtypes in estrogen receptor-positive breast carcinomas through genomic grade . J Clin Oncol 25 , 1239 – 1246 ( 2007 ). OpenUrl Abstract / FREE Full Text 31. Loi , S. et al. Predicting prognosis using molecular profiling in estrogen receptor-positive breast cancer treated with tamoxifen . BMC Genomics 9 , 239 ( 2008 ). OpenUrl CrossRef PubMed 32. ↡ Poudel , P. , Nyamundanda , G. , Patil , Y. , Cheang , M. C. U. & Sadanandam , A. Heterocellular gene signatures reveal luminal-A breast cancer heterogeneity and differential therapeutic responses . npj Breast Cancer 5 , 1 – 10 ( 2019 ). OpenUrl PubMed 33. ↡ Stopsack , K. H. et al. Extent, impact, and mitigation of batch effects in tumor biomarker studies using tissue microarrays . eLife 10 , e71265 . View the discussion thread. Back to top Previous Next Posted December 26, 2024. Download PDF Data/Code Email Thank you for your interest in spreading the word about medRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following The significance of molecular heterogeneity in breast cancer batch correction and dataset integration Message Subject (Your Name) has forwarded a page to you from medRxiv Message Body (Your Name) thought you would like to see this page from the medRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share The significance of molecular heterogeneity in breast cancer batch correction and dataset integration Nicholas Moir , Dominic A. Pearce , Simon P. Langdon , T. Ian Simpson medRxiv 2024.12.22.24319524; doi: https://doi.org/10.1101/2024.12.22.24319524 Share This Article: Copy Citation Tools The significance of molecular heterogeneity in breast cancer batch correction and dataset integration Nicholas Moir , Dominic A. Pearce , Simon P. Langdon , T. Ian Simpson medRxiv 2024.12.22.24319524; doi: https://doi.org/10.1101/2024.12.22.24319524 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Oncology Subject Areas All Articles Addiction Medicine (573) Allergy and Immunology (865) Anesthesia (304) Cardiovascular Medicine (4457) Dentistry and Oral Medicine (445) Dermatology (383) Emergency Medicine (610) Endocrinology (including Diabetes Mellitus and Metabolic Disease) (1517) Epidemiology (15244) Forensic Medicine (30) Gastroenterology (1132) Genetic and Genomic Medicine (6620) Geriatric Medicine (669) Health Economics (1002) Health Informatics (4557) Health Policy (1372) Health Systems and Quality Improvement (1615) Hematology (543) HIV/AIDS (1272) Infectious Diseases (except HIV/AIDS) (15936) Intensive Care and Critical Care Medicine (1106) Medical Education (624) Medical Ethics (147) Nephrology (670) Neurology (6635) Nursing (346) Nutrition (999) Obstetrics and Gynecology (1148) Occupational and Environmental Health (957) Oncology (3348) Ophthalmology (980) Orthopedics (369) Otolaryngology (421) Pain Medicine (436) Palliative Medicine (130) Pathology (665) Pediatrics (1696) Pharmacology and Therapeutics (693) Primary Care Research (714) Psychiatry and Clinical Psychology (5463) Public and Global Health (9257) Radiology and Imaging (2210) Rehabilitation Medicine and Physical Therapy (1371) Respiratory Medicine (1198) Rheumatology (598) Sexual and Reproductive Health (716) Sports Medicine (532) Surgery (714) Toxicology (99) Transplantation (289) Urology (265) (function(){function c(){var b=a.contentDocument||a.contentWindow.document;if(b){var d=b.createElement('script');d.innerHTML="window.__CF$cv$params={r:'a035d02699e452ad',t:'MTc4MDA2MTAwMA=='};var a=document.createElement('script');a.src='/cdn-cgi/challenge-platform/scripts/jsd/main.js';document.getElementsByTagName('head')[0].appendChild(a);";b.getElementsByTagName('head')[0].appendChild(d)}}if(document.body){var a=document.createElement('iframe');a.height=1;a.width=1;a.style.position='absolute';a.style.top=0;a.style.left=0;a.style.border='none';a.style.visibility='hidden';document.body.appendChild(a);if('loading'!==document.readyState)c();else if(window.addEventListener)document.addEventListener('DOMContentLoaded',c);else{var e=document.onreadystatechange||function(){};document.onreadystatechange=function(b){e(b);'loading'!==document.readyState&&(document.onreadystatechange=e,c())}}}})();

Text is read by the "Ask this paper" AI Q&A widget below. Extraction quality varies by source β€” PMC NXML preserves structure cleanly, OA-HTML may include some navigation residue, and OA-PDF can have broken hyphenation. The publisher copy (via DOI) is the canonical version.

My notes (saved in your browser only)

βš™ Ask this paper AI returns verbatim quotes from the full text Β· source: preprint-html β“˜

Answers must be backed by verbatim quotes from this paper's full text. Hallucinated quotes are dropped automatically; if no verbatim passage answers the question, we say so. How this works

Citation neighborhood (no data yet)

We don't have any in-corpus citations linked to this paper yet. This is a recent paper (2024) β€” citers typically take a year or two to land, and the OpenAlex reference graph may still be filling in.

Source provenance

europepmc
last seen: 2026-05-20T01:45:00.602351+00:00