Full text
36,165 characters
· extracted from
preprint-html
· click to expand
BASSL-MI: Batch-Agnostic Self-Supervised Learning Uncovers Clinically Relevant Tumor Niches in Multiplexed Imaging | bioRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-M677548'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search New Results BASSL-MI: Batch-Agnostic Self-Supervised Learning Uncovers Clinically Relevant Tumor Niches in Multiplexed Imaging View ORCID Profile Alexander Lin , View ORCID Profile Shunxing Bao , View ORCID Profile Ken Lau , View ORCID Profile Simon Vandekar , View ORCID Profile Daniel Moyer , Qi Liu , View ORCID Profile Siyuan Ma doi: https://doi.org/10.1101/2025.11.04.686632 Alexander Lin 1 Department of Computer Science, Vanderbilt University Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Alexander Lin Shunxing Bao 2 Department of Electrical and Computer Engineering, Vanderbilt University Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Shunxing Bao Ken Lau 3 Department of Cell and Developmental Biology, Vanderbilt University Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Ken Lau Simon Vandekar 4 Department of Biostatistics, Vanderbilt University Medical Center Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Simon Vandekar Daniel Moyer 1 Department of Computer Science, Vanderbilt University Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Daniel Moyer For correspondence: daniel.moyer{at}vanderbilt.edu qi.liu{at}vumc.org siyuan.ma{at}vumc.org Qi Liu 4 Department of Biostatistics, Vanderbilt University Medical Center Find this author on Google Scholar Find this author on PubMed Search for this author on this site For correspondence: daniel.moyer{at}vanderbilt.edu qi.liu{at}vumc.org siyuan.ma{at}vumc.org Siyuan Ma 4 Department of Biostatistics, Vanderbilt University Medical Center Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Siyuan Ma For correspondence: daniel.moyer{at}vanderbilt.edu qi.liu{at}vumc.org siyuan.ma{at}vumc.org Abstract Full Text Info/History Metrics Preview PDF ABSTRACT Multiplexed imaging enables rich, high-resolution characterization of the tumor microenvironment but relies on labor-intensive and error-prone cell segmentation and phenotyping pipelines. We present BASSL-MI, a batch-agnostic, self-supervised framework for discovering tissue niches directly from multiplexed imaging data. BASSL-MI operates directly on image patches, eliminating the need for explicit cell segmentation while mitigating image-, sample-, or batch-specific artifacts. Built on a modified contrastive block disentanglement architecture, BASSL-MI learns dual latent representations that separate biologically informative features from batch-dependent factors through spatially guided augmentations and batch-invariance objectives. Applied to a 56-marker colorectal cancer CODEX dataset, BASSL-MI-trained embeddings markedly reduce image-specific variability and recover biologically interpretable spatial niches. Notably, it uncovers CD20–rich follicular regions associated with improved survival, outperforming published findings from cell segmentation-driven clustering. This work demonstrates that self-supervised, patch-based learning can capture clinically relevant spatial organization within tumor microenvironments, advancing toward automated, non-cell-based analysis of multiplexed imaging data. 1. INTRODUCTION Recent advances in multiplexed imaging technologies have transformed our ability to characterize tissue architecture at cellular and subcellular resolution [ 1 , 2 , 3 ]. Unlike traditional immunohistochemistry (IHC), which measures only a handful of protein markers, multiplexed imaging simultaneously quantifies dozens of protein channels [ 4 ]. Platforms such as CO-Detection by Indexing (CODEX) [ 1 ], Tissue-based Cyclic Immunofluorescence (t-CyCIF) [ 2 ], and Imaging Mass Cytometry (IMC) [ 3 ] have been successfully applied to diverse health conditions and organs to map protein expression and delineate tissue “niches” – biologically and clinically meaningful organization of cells and protein expression, which have been associated with disease phenotype, risk, and prognosis [ 1 , 5 , 6 ] Despite these advances, characterizing tissue niches with multiplexed imaging remains challenging because typical workflows depend on error-prone and laborious steps. Analyses commonly begin with cell segmentation, where individual cells are delineated using nuclear and membrane markers. This is complicated by irregularities in cell size and shape (common in tumor tissues) or signal bleed-through [ 7 ]. The next step is cell phenotyping, where manual marker gating [ 8 ] or clustering approaches adapted from existing single-cell methods [ 9 ] are used to assign each cell to a biologically meaningful class. Both methods are labor intensive, require subjective determination of cell phenotypes, and are sensitive to imaging noise and spatial artifacts [ 10 ]. Errors in these foundational steps can propagate through subsequent analyses and distort biological conclusions [ 1 , 7 ], underscoring the need for more robust, automated workflows. New developments in self-supervised learning (SSL) offer approaches that learn informative embeddings of unlabeled data via pretext tasks, eliminating the need for expert-level annotations [ 11 , 12 ]. When applied to multiplexed imaging, these approaches can operate on small, multi-channel “patches” cut from larger images [ 13 , 14 ]. The resulting patch embeddings naturally represent local microenvironments and can be directly used for downstream analysis, bypassing cell segmentation and phenotyping. A key technical challenge, however, is image-specific effects – technical variability arising from differences in e.g. staining, imaging protocols, and sample handling – which, if not corrected for, will dominate patches’ learned representations and obscure true biological signals [ 15 , 16 ]. Preliminary applications of SSL to multiplexed imaging have shown early success: Atarsaikhan et al. [ 13 ] used hierarchical SSL to learn multiscale tissue representations that captured prognostically relevant niches, while Su et al. [ 14 ] proposed AdvDINO, a domain-adversarial framework that mitigated slide-specific biases and identified clinically meaningful clusters in lung cancer multiplexed immunofluorescence (mIF) images. However, these methods either do not explicitly address batch effects or rely on adversarial training, which increases computational complexity. Recent work suggests that disentangled or contrastive objectives can achieve batch invariance without adversarial optimization [ 17 , 18 ], motivating non-adversarial SSL for robust and biologically meaningful representation learning in multiplexed imaging. We propose BASSL-MI ( B atch- A gnostic S elf- S upervised L earning for M ultiplexed Imaging), a framework for clustering tissue niches in multiplexed imaging data. Briefly, our contributions include: Streamlined tissue niche discovery with SSL, operating directly on image patches without cell segmentation/phenotyping steps. Explicit correction of batch effects in multiplexed imaging without adversarial learning, unlike existing methods [ 13 , 14 ]. As validated on a colorectal cancer (CRC) study [ 19 ], our learned tumor niches are agnostic to batch effects and discern patient survival, outperforming published, cell-based findings. 2. METHODS 2.1. Dataset and processing The dataset used in this work consists of 140 tissue samples collected from 35 CRC patients, each annotated with clinical and demographic metadata such as gender, age, and overall survival. To gain insight into the tumor microenvironment, each tissue sample was imaged with a 56-marker CODEX, encompassing immune, checkpoint, membrane, and other features [ 19 ]. Seventeen of the patients exhibited Crohn’s-like reaction (CLR), characterized by the formation of tertiary lymphoid structures (TLSs) – organized immune aggregates that have been associated with improved prognosis [ 20 ]. In contrast, the remaining eighteen patients showed diffuse inflammatory infiltration (DII), where immune cells are dispersed throughout the tissue without forming organized structures. This difference in immune spatial architecture has important biological and clinical implications: patients with CLR and TLS formation typically demonstrate significantly better overall survival compared to those with DII, who lack such organized immune responses [ 19 , 20 ]. Twenty-three of the imaged markers (including HOECHST for nuclei) were selected and utilized in this study, focusing on those that defined cell types and other structural features. To mitigate broader image-specific batch effects, pixel intensity values for these markers were log normalized on a per-channel, per-image basis, following [ 16 ]. 128 × 128 patches were cut from each image using a sliding window of step size 32 and fed to the self-supervised models for training; the patch dimensions approximately agree with tumor niches as considered in the original publication [ 19 ]. 2.2. Learning framework BASSL-MI takes inspiration from the supervised contrastive block disentanglement (SCBD) algorithm [ 18 ] for SSL. The goal of SCBD is to learn representations of an input x that are informative of labels c while invariant to nuisance factors b . We learn two disjoint embeddings: one encodes information about an entity of interest ( z c ); the other encodes information about the unwanted factor ( z b ). We modify SCBD by generalizing the supervision on c . In our formulation, c no longer corresponds to a specific class label but instead corresponds to different augmentations (see below) of the same or similar underlying input x . This preserves the unsupervised nature of tissue niche discovery while removing unwanted batch effects. Our method is diagrammed in Fig. 1 . Given input patches x , we learn two encoder networks, f c and f b , that transform x into intermediate representations z c and z b , respectively. Download figure Open in new tab Fig. 1. Overview of BASSL-MI. Each anchor patch is a small multi-channel region from a multiplexed image. Augmented neighboring patches serve as class positives (similar biology), same-slide patches as batch positives, and patches from other slides as batch negatives. Two encoders learn separate representations: f c for biological content ( z c ) and f b for batch variation ( z b ). The decoder reconstructs the input from both latents, while contrastive and invariance losses ensure z c captures biological instead of batch signals. The red color on z c indicates this representation is used for downstream analysis. We also learn two networks, h c and h b , that map the intermediate z representations to their projections, p c , and p b , which are normalized to the unit hyper-sphere and are used for loss function calculations. During the decoding process, z c and z b are concatenated and passed to the decoder g to produce a reconstruction of the input . Our objective consists of four components: ℒ recon , ℒ C , ℒ B , ℒ inv . Reconstruction loss ℒ recon is the summed mean-squared error between the original input and reconstruction over minibatch ℬ: Next, we define the InfoNCE loss [ 21 ], which is a major component of the remaining three objectives. where y is the target factor, τ is a temperature hyperparameter, i is the current sample (anchor) index, and j is the positive sample index. The contrastive loss terms ℒ C and ℒ B are direct applications of the InfoNCE loss as applied to p b and p c , respectively: Here, P c ( i ) and P b ( i ) denote the set of positive samples relative to the anchor i for factors c and b , respectively. The final term ℒ inv represents the batch invariance loss. The goal of this objective is to make the z c representations invariant to the external factor b . It performs this optimization by matching the InfoNCE losses on factor c between samples within the same batch as anchor i and samples from other batches. where P b ( i ) is from above and N b ( i ) = ℬ \ { P b ( i ) ∪ i }. From a classification standpoint, this can be seen as minimizing the classification accuracy at the batch level. The overall objective for this framework can be written as where β C , β B , and β inv are prespecified hyperparameters. We implement f c and f b as convolutional neural networks followed by fully connected multilayer perceptrons (MLPs) that take flattened convolutional representations and map them to z c and z b . h c and h b are fully connected MLPs, mapping z c and z b to p c and p b . The decoder g is a transpose convolutional network that maps the concatenation of z c and z b back to the original patch domain. To complete the contrastive framework, we utilize a spatial heuristic to define positive samples (class positives). For each anchor patch, we select a spatially adjacent patch (or the patch itself) as a positive sample, under the assumption that neighboring regions in tissue share similar biological characteristics. We then apply a series of augmentations – random rotations, horizontal and vertical reflections, and small amounts of Gaussian noise – to both the anchor and positive patches. This spatially informed augmentation strategy encourages the model to learn representations invariant to technical noise and uninformative imaging factors (such as orientation or illumination), while capturing meaningful biological variation. The external factor b that we seek to establish invariance over is the image the patches come from. Previous literature has described the presence of image-specific variations resulting from technical noise, not biological patterns. Thus, we utilize this framework to create batch-agnostic representations z c and instead store batch-specific information within z b . Patches from the same image serve as batch positives, while patches from other images serve as batch negatives. 2.3. Spatially-aware clustering After model training was complete, we extracted the batch-agnostic latent variables ( z c ) for all patches in the dataset and performed Leiden clustering [ 22 ] on these representations to identify groups of spatially and phenotypically similar tissue regions. Following this, we applied spatial smoothing across neighboring patches prior to analysis to reduce noise from isolated misclustered areas. We then examined marker expression enrichment within each cluster to characterize their underlying biological composition and computed the average cluster frequency per patient to compare the prevalence of specific clusters between the CLR and DII patient subgroups. 2.4. Comparison considerations Of the two existing SSL applications in multiplexed imaging, [ 13 ] does not account for image-specific effects and corresponds to a non-batch-adjusted baseline. [ 14 ] does not have public code available. As such, we focus our batch correction comparison on non-batch-adjusted baseline and an ablation of our invariance loss component. We further conduct clinical relevance comparisons of BASSL-MI-identified tumor niches and those based on cell segmentation, phenotyping, and composition clustering (reported in the original publication [ 19 ]). Cell-based niche analyses remain the most common analytical approach to date, and comparisons against such cell-based findings provide application baselines for BASSL-MI. Specifically, each patch was assigned a baseline niche label based on annotations from [ 19 ], where each cell’s local microenvironment – defined by the composition of surrounding cells – was used to identify its tumor niche. We assigned each patch the plurality tumor niche label among the cells within that patch, breaking ties arbitrarily. These labels were used as a baseline to compare against BASSL-MI-derived niches, in terms of their clinical correlation with patient survival outcome. 3. RESULTS 3.1. Reduction of image-specific batch effects BASSL-MI substantially reduces image-specific batch effects in the learned latent representations. As expected, strong batch effects were initially observed across patient images. To overcome this, we incorporated several batch-correction measures into the SSL framework, including blocking the latent space and an explicit batch-invariance loss objective (see Methods for details). Conceptually, the goal of this learning framework is for the model to disentangle biologically relevant features from unwanted variation. For comparison, we trained a baseline model without the batch invariance loss or partition of the latent space, similar in spirit to [ 13 ]. As visualized in Fig. 2(a) , the resulting embeddings showed clear batch-specific segregation, with patches clustering primarily by batch in the UMAP. Introduction of the latent space partition disperses the learned representations across the UMAP embedding space, as visualized in panel (b); however, batch-specific artifacts still appear. When latent space blocking is combined with the batch invariance loss, the embeddings become further interleaved across batches ( Fig. 2(c) ), demonstrating increased suppression of batch-driven structure. This trend is consistent with the statistics reported in Table 1 , where the dependence of the latent space on batch ( R 2 ↓) decreases from 0.327 (no correction) to 0.285 (latent space blocking) and further to 0.160 (with invariance loss and blocking), confirming that the combined strategy most effectively mitigates batch effects. View this table: View inline View popup Download powerpoint Table 1. Ablation Test: Dependency of latent space on image, measured as R 2 from regression of latent dimensions on image. Download figure Open in new tab Fig. 2. BASSL-MI reduces batch effects in learned representations. Each point represents a patch, colored by its source image (batch). (a) Without correction, embeddings cluster strongly by image, indicating pronounced batch effects. (b) Latent space blocking reduces this effect, though some batch grouping remains (red arrows). (c) Combining latent space blocking with the invariance loss produces well-mixed embeddings, demonstrating that BASSL-MI performs effective batch correction. 3.2. Learned latent representations are biologically and clinically informative BASSL-MI learned tumor niches more truthfully represent protein channels’ spatial organization compared to the published baseline learned through segmented and phenotyped cells ( Fig. 3 ). Panel (a) shows two representative images. The left column shows raw marker expression on the tissue; the middle column shows the distribution of BASSL-MI niches; and the right column shows the distribution of the baseline niche labels derived from [ 19 ] (cell-based clusters). Visually, for the top sample, BASSL-MI clusters define the major follicular region in the center of the image. The four epithelial regions on the left, which overlap with the MUC-1 marker in the original image, are identified in the four red globular regions from BASSL-MI latent representations (cluster 5). The yellow niche (cluster 7) corresponds with the large smooth muscle region at the top of the image, and the magenta niches (cluster 10) correspond with increased expression of CD56, a marker for NK cells. In contrast, the tumor niches from [ 19 ] capture some of these regions, but they notably miss the NK cell-enriched regions and some of the epithelial regions. For the second sample, we see increased CD20 expression (in green) in the top of the original image. This sample is from a CLR patient, indicating a likely representation of a TLS. BASSL-MI representations successfully recaptures this (cluster 20), whereas the baseline labels classify that region as T-cell enriched. Download figure Open in new tab Fig. 3. Biological and clinical relevance of BASSL-MI tumor niches. (a) Representative visualization of learned niche labels demonstrate better agreement between BASSL-MI-identified niches and actual protein expression, compared to published, cell-based labels [ 19 ]. (b) The BASSL-MI Follicle niche is enriched in CLR compared to DII patients, aligning with expectations. (c) Frequency of BASSL-MI Follicle niches is associated with overall survival. Kaplan-Meier survival curves represent patient stratification using the niche frequency’s median. (d) When separately associated with patient survival with Cox proportional hazard regression, frequency of BASSL-MI follicle had a significant hazard ratio whereas cell-based follicle did not. (e) In joint likelihood ratio (LRT) analysis of survival versus the two follicle labels, BASSL-MI follicle had a higher χ 2 statistic, indicating a stronger signal for survival compared to cell-based follicle when mutually adjusted. The BASSL-MI-identified B-cell niche has clinically meaningful associations with patient phenotype and outcome ( Fig. 3 (b,c) ). Our cluster 20 exhibited notably high expression of CD20, a canonical B-cell marker, suggesting correspondence to B-cell–rich follicular regions (present in CLR patients, largely absent in DII patients). We thus term it as “BASSL-MI follicle” and further examine its validity and clinical relevance. First, in terms of the proportion of the niche in each patient, it was significantly enriched in CLR patients ( p adj = 0.016, Benjamini–Hochberg correction [ 23 ], Fig. 3 (b) ), congruent with expectations. Furthermore, Kaplan–Meier survival analysis based on stratification by median niche frequency (high versus low) demonstrated a significant difference in overall survival ( p = 0.031, Fig. 3 (c) ). These findings agree with [ 19 ] and validate BASSL-MI’s ability to identify distinct spatial features of CLR tissues and their associations with improved clinical outcomes. Furthermore, BASSL-MI-identified follicles have improved discrimination of patient survival compared to baseline, cell-based follicles ( Fig. 3 (d , e) ). We fit marginal Cox proportional hazard regressions associating patient survival with the frequency of either BASSL-MI follicles or cell-based follicles. The hazard ratio is significant for BASSL-MI follicle frequency ( p = 0.034), but non-significant for cell-based follicle ( Fig. 3 (d) ). We further conduct likelihood ratio testing (LRT) by comparing the above marginal regressions against a mutually adjusted Cox regression with both terms included. While neither is significant, removing BASSL-MI follicle frequencies from the full model yields a higher χ 2 statistic ( p = 0.16), suggesting a trend that BASSL-MI follicles are more informative for survival ( Fig. 3 (e) ). These findings suggest that follicle structures identified by BASSL-MI have closer associations with patient survival compared to baseline cell-based labeling, likely due to BASSL-MI’s more truthful alignment between identified tumor niches and actual protein expression in the original samples. 4. CONCLUSION Existing multiplexed imaging analysis relies on manual and sensitive algorithms for cell segmentation and phenotyping, where errors can propagate and potentially distort conclusions. Moreover, these approaches reduce rich spatial information to the single-cell level, losing critical tissue contexts. We present BASSL-MI, a batch-agnostic, self-supervised framework that does not require cell segmentation/phenotyping, corrects for batch effects inherent to multiplexed imaging, and produces biologically interpretable niches. It represents one of the first steps toward non-cell-based multiplexed imaging analysis. Future extensions will explore more disease types, semi-supervised strategies to incorporate human annotation, and higher-order spatial interactions between tissue niches. 5. COMPLIANCE WITH ETHICAL STANDARDS This work analyzes public, de-identified tissue section imaging data and represents secondary use of existing, non-identifiable research data. No new human subjects were recruited, and no personally identifiable information was accessed or analyzed. As such, it was exempt from institutional review board (IRB) oversight. 6. ACKNOWLEDGMENTS This work is partially supported by the Vanderbilt-Ingram Cancer Center GI SPORE Career Enhancement Program (parent grant NIH U2CCA233291). The authors would like to acknowledge input from Dr. Lianrui Zuo and Dr. Matthew Berger in supporting this work. Funder Information Declared National Institutes of Health , U2CCA233291 7. REFERENCES [1]. ↵ Yury Goltsev , Nikolay Samusik , Julia Kennedy-Darling , Salil Bhate , Matthew Hale , Gustavo Vazquez , et al. , “ Deep Profiling of Mouse Splenic Architecture with CODEX Multiplexed Imaging ,” Cell , vol. 174 , no. 4 , pp. 968 – 981 .e15, Aug . 2018 . OpenUrl CrossRef PubMed [2]. ↵ Jia-Ren Lin , Benjamin Izar , Shu Wang , Clarence Yapp , Shaolin Mei , Parin M Shah , et al. , “ Highly multiplexed immunofluo-rescence imaging of human tissues and tumors using t-cycif and conventional optical microscopes ,” eLife , vol. 7 , pp. e31657 , jul 2018 . OpenUrl CrossRef PubMed [3]. ↵ Charlotte Giesen , Hao AO Wang , Denis Schapiro , Nevena Zivanovic , Andrea Jacobs , Bodo Hattendorf , et al. , “ Highly multiplexed imaging of tumor tissues with subcellular resolution by mass cytometry ,” Nature methods , vol. 11 , no. 4 , pp. 417 – 422 , 2014 . OpenUrl PubMed [4]. ↵ Alina Bollhagen and Bernd Bodenmiller , “ Highly Multiplexed Tissue Imaging in Precision Oncology and Translational Cancer Research ,” Cancer Discovery , vol. 14 , no. 11 , pp. 2071 – 2088 , Nov . 2024 . OpenUrl CrossRef PubMed [5]. ↵ Ayano Kondo , Siyuan Ma , Michelle YY Lee , Vivian Ortiz , Daniel Traum , Jonathan Schug , et al. , “ Highly multiplexed image analysis of intestinal tissue sections in patients with inflammatory bowel disease ,” Gastroenterology , vol. 161 , no. 6 , pp. 1940 – 1952 , 2021 . OpenUrl CrossRef PubMed [6]. ↵ Siyuan Ma , Nawras W Habash , Mrunal K Dehankar , Nidhi Jalan-Sakrikar , Shawna A Cooper , Abid A Anwar , et al. , “ Congestion enriches intra-hepatic macrophages through reverse zonation of cxcl9 in liver sinusoidal endothelial cells ,” Cellular and Molecular Gastroenterology and Hepatology , vol. 19 , no. 7 , pp. 101475 , 2025 . OpenUrl [7]. ↵ Matthias Bruhns , Jan T. Schleicher , Maximilian Wirth , Marcello Zago , Sepideh Babaei , and Manfred Claassen , “ Effects of segmentation errors on downstream-analysis in highlymultiplexed tissue imaging ,” PLOS Computational Biology , vol. 21 , no. 9 , pp. e1013350 , Sept . 2025 . OpenUrl [8]. ↵ Bob Chen , Scurrah Cherie’R , Eliot T McKinley , Alan J Simmons , Marisol A Ramirez-Solano , Xiangzhu Zhu , et al. , “ Differential pre-malignant programs and microenvironment chart distinct paths to malignancy in human colorectal polyps ,” Cell , vol. 184 , no. 26 , pp. 6262 – 6280 , 2021 . OpenUrl CrossRef PubMed [9]. ↵ Vladimir Yu Kiselev , Tallulah S Andrews , and Martin Hemberg , “ Challenges in unsupervised clustering of single-cell rna-seq data ,” Nature Reviews Genetics , vol. 20 , no. 5 , pp. 273 – 282 , 2019 . OpenUrl CrossRef PubMed [10]. ↵ Jiangmei Xiong , Harsimran Kaur , Cody N Heiser , Eliot T McKinley , Joseph T Roland , Robert J Coffey , et al. , “ Gammagater: semi-automated marker gating for single-cell multiplexed imaging ,” Bioinformatics , vol. 40 , no. 6 , pp. btae356 , 2024 . OpenUrl PubMed [11]. ↵ Ting Chen , Simon Kornblith , Mohammad Norouzi , and Geoffrey Hinton , “ A simple framework for contrastive learning of visual representations ,” in Proceedings of the 37th International Conference on Machine Learning. 2020, ICML’20, JMLR.org . [12]. ↵ Maxime Oquab , Timotheé Darcet Theó Moutakanni Huy Vo , Marc Szafraniec , Vasil Khalidov , et al. , “ Dinov2: Learning robust visual features without supervision ,” 2024 . [13]. ↵ Gantugs Atarsaikhan , Isabel Mogollon , Katja Välimäki , iCAN , Tuomas Mirtti , Teijo Pellinen , et al. , “ Self-supervised learning enables unbiased patient characterization from multiplexed microscopy images ,” bioRxiv , 2025 . [14]. ↵ Stella Su , Marc Harary , Scott J. Rodig , and William Lotter , “ AdvDINO: Domain-Adversarial Self-Supervised Representation Learning for Spatial Proteomics ,” Aug . 2025 , arxiv: 2508.04955 [cs]. [15]. ↵ Daniel Moyer , Greg Ver Steeg , Chantal MW Tax , and Paul M Thompson , “ Scanner invariant representations for diffusion mri harmonization ,” Magnetic resonance in medicine , vol. 84 , no. 4 , pp. 2174 – 2189 , 2020 . OpenUrl CrossRef PubMed [16]. ↵ Coleman R Harris , Eliot T McKinley , Joseph T Roland , Qi Liu , Martha J Shrubsole , Ken S Lau , et al. , “ Quantifying and correcting slide-to-slide variation in multiplexed immunofluorescence images ,” Bioinformatics , vol. 38 , no. 6 , pp. 1700 – 1707 , Mar . 2022 . OpenUrl PubMed [17]. ↵ Daniel Moyer , Shuyang Gao , Rob Brekelmans , Aram Galstyan , and Greg Ver Steeg , “ Invariant representations without adversarial training ,” Advances in neural information processing systems , vol. 31 , 2018 . [18]. ↵ Taro Makino , Ji Won Park , Natasa Tagasovska , Takamasa Kudo , Paula Coelho , Jan-Christian Huetter , et al. , “Supervised Contrastive Block Disentanglement,” Feb . 2025 , arxiv: 2502.07281 [cs]. [19]. ↵ Christian M. Schürch , Salil S. Bhate , Graham L. Barlow , Darci J. Phillips , Luca Noti , Inti Zlobec , et al. , “ Coordinated Cellular Neighborhoods Orchestrate Antitumoral Immunity at the Colorectal Cancer Invasive Front ,” Cell , vol. 182 , no. 5 , pp. 1341 – 1359 .e19, Sept . 2020 . OpenUrl CrossRef PubMed [20]. ↵ David M. Graham and Henry D. Appelman , “ Crohn’s-like lymphoid reaction and colorectal carcinoma: a potential histologic prognosticator .,” Modern pathology : an official journal of the United States and Canadian Academy of Pathology, Inc , vol. 3 3 , pp. 332 – 5 , 1990 . OpenUrl [21]. ↵ Aaron van den Oord , Yazhe Li , and Oriol Vinyals , “ Representation Learning with Contrastive Predictive Coding ,” Jan . 2019 , arxiv: 1807.03748 [cs]. [22]. ↵ V. A. Traag , L. Waltman , and N. J. van Eck , “ From louvain to leiden: guaranteeing well-connected communities ,” Scientific Reports , vol. 9 , no. 1 , Mar . 2019 . [23]. ↵ Yoav Benjamini and Yosef Hochberg , “ Controlling the false discovery rate: A practical and powerful approach to multiple testing ,” Journal of the Royal Statistical Society. Series B (Methodological) , vol. 57 , no. 1 , pp. 289 – 300 , 1995 . OpenUrl CrossRef PubMed Web of Science View the discussion thread. Back to top Previous Next Posted November 05, 2025. Download PDF Email Thank you for your interest in spreading the word about bioRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following BASSL-MI: Batch-Agnostic Self-Supervised Learning Uncovers Clinically Relevant Tumor Niches in Multiplexed Imaging Message Subject (Your Name) has forwarded a page to you from bioRxiv Message Body (Your Name) thought you would like to see this page from the bioRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share BASSL-MI: Batch-Agnostic Self-Supervised Learning Uncovers Clinically Relevant Tumor Niches in Multiplexed Imaging Alexander Lin , Shunxing Bao , Ken Lau , Simon Vandekar , Daniel Moyer , Qi Liu , Siyuan Ma bioRxiv 2025.11.04.686632; doi: https://doi.org/10.1101/2025.11.04.686632 Share This Article: Copy Citation Tools BASSL-MI: Batch-Agnostic Self-Supervised Learning Uncovers Clinically Relevant Tumor Niches in Multiplexed Imaging Alexander Lin , Shunxing Bao , Ken Lau , Simon Vandekar , Daniel Moyer , Qi Liu , Siyuan Ma bioRxiv 2025.11.04.686632; doi: https://doi.org/10.1101/2025.11.04.686632 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Bioinformatics Subject Areas All Articles Animal Behavior and Cognition (7637) Biochemistry (17705) Bioengineering (13899) Bioinformatics (41968) Biophysics (21460) Cancer Biology (18603) Cell Biology (25526) Clinical Trials (138) Developmental Biology (13385) Ecology (19909) Epidemiology (2067) Evolutionary Biology (24326) Genetics (15614) Genomics (22513) Immunology (17741) Microbiology (40423) Molecular Biology (17193) Neuroscience (88645) Paleontology (667) Pathology (2835) Pharmacology and Toxicology (4825) Physiology (7647) Plant Biology (15160) Scientific Communication and Education (2046) Synthetic Biology (4302) Systems Biology (9825) Zoology (2271)
Text is read by the "Ask this paper" AI Q&A widget below.
Extraction quality varies by source — PMC NXML preserves structure
cleanly, OA-HTML may include some navigation residue, and OA-PDF can
have broken hyphenation. The publisher copy
(via DOI)
is the canonical version.