Full text
30,274 characters
· extracted from
preprint-html
· click to expand
FADVI: disentangled representation learning for robust integration of single-cell and spatial omics data | bioRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-M677548'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search New Results FADVI: disentangled representation learning for robust integration of single-cell and spatial omics data View ORCID Profile Wendao Liu , Gang Qu , Lukas M. Simon , View ORCID Profile Fabian J. Theis , View ORCID Profile Zhongming Zhao doi: https://doi.org/10.1101/2025.11.03.683998 Wendao Liu 1 The University of Texas MD Anderson Cancer Center UTHealth Houston Graduate School of Biomedical Sciences , Houston, TX, USA 2 Center for Precision Health, McWilliams School of Biomedical Informatics, The University of Texas Health Science Center at Houston , Houston, TX, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Wendao Liu For correspondence: wendao.liu{at}uth.tmc.edu zhongming.zhao{at}uth.tmc.edu Gang Qu 2 Center for Precision Health, McWilliams School of Biomedical Informatics, The University of Texas Health Science Center at Houston , Houston, TX, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site Lukas M. Simon 3 Therapeutic Innovation Center, Baylor College of Medicine , Houston, TX, USA 4 Verna & Marrs McLean Department of Biochemistry and Molecular Pharmacology, Baylor College of Medicine , Houston, TX, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site Fabian J. Theis 5 Institute of Computational Biology, Helmholtz Zentrum München , Neuherberg, Germany 6 TUM School of Life Sciences Weihenstephan, Technical University of Munich , Freising, Germany 7 Department of Mathematics, Technical University of Munich , Garching, Germany Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Fabian J. Theis Zhongming Zhao 1 The University of Texas MD Anderson Cancer Center UTHealth Houston Graduate School of Biomedical Sciences , Houston, TX, USA 2 Center for Precision Health, McWilliams School of Biomedical Informatics, The University of Texas Health Science Center at Houston , Houston, TX, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Zhongming Zhao For correspondence: wendao.liu{at}uth.tmc.edu zhongming.zhao{at}uth.tmc.edu Abstract Full Text Info/History Metrics Supplementary material Preview PDF Abstract Integrating single-cell and spatial omics data remains challenging due to strong batch effects across experiments and platforms. Existing methods focus on minimizing these effects but cannot disentangle technical variation from true biological signals. Here, we present FADVI, a variational autoencoder framework partitioning the latent space into batch-specific, label-related, and residual subspaces. By combining supervised classification, adversarial training, and cross-covariance penalty, FADVI enforces independent representations that preserve biological variation while correcting batch effects. Benchmarking across scRNA-seq, scATAC-seq, and high-resolution spatial transcriptomics datasets, FADVI consistently outperformed state-of-the-art integration methods. FADVI also enables feature attribution, revealing genes associated with cell type identity and batch variation. Together, these results demonstrate that FADVI provides robust, interpretable integration for large-scale single-cell and spatial omics data, offering a powerful framework for downstream analysis and discovery. Main With the rapid advancement and widespread adoption of single-cell sequencing and spatial transcriptomics (ST), the scale and complexity of datasets have grown dramatically. Batch effects, caused by unwanted technical variation, arise from sources such as experimental replicates or profiling technologies. These effects present a major challenge for processing atlas-scale datasets and integrating high-resolution ST with single-cell RNA sequencing (scRNA-seq) data. Numerous computational methods have been developed for single-cell batch correction, and comprehensive benchmarking studies have identified top-performing approaches across diverse datasets 1 , 2 . Among them, scANVI 3 , a semi-supervised variational autoencoder (VAE)-based method incorporating cell type labels, has consistently achieved state-of-the-art performance, outperforming unsupervised methods. Unlike existing methods that primarily aim to minimize batch effects, our goal was to disentangle batch variation from biological signals, as disentanglement may improve single-cell data integration, annotation, generation, and counterfactual prediction 4 - 6 . To this end, we developed FActor Disentanglement Variational Inference (FADVI), a VAE-based framework that partitions the latent space into three distinct subspaces: batch-specific, label-related, and residual variation ( Fig. 1a ). The encoder maps input into three latent factors, which are then combined by the decoder to reconstruct the original data. To ensure that each factor captures only the intended information, we introduce supervised classification heads, adversarial networks with gradient reversal layers, and a cross-covariance penalty. The supervised heads align the batch and label factors with input observations, while the adversarial heads prevent leakage of unintended information across factors. The cross-covariance penalty further enforces statistical independence among the latent subspaces. This design enables FADVI to generate disentangled representations that support integration, clustering, visualization, and label projection, while enhancing interpretability and robustness. Similar to scANVI, FADVI follows a semi-supervised paradigm, capable of integrating both labeled and unlabeled data with shared cell type composition. Download figure Open in new tab Fig. 1. FADVI disentangles batch and biological variation for atlas-scale scRNA-seq data integration. a, Schematic of FADVI architecture. b, FADVI shows higher integration scores with competing methods across six scRNA-seq datasets. FADVI(semi): semi-supervised training with masked cell type labels in half batches. FADVI(un): unsupervised training without cell type labels. c, Individual and aggregated metrics for evaluating integration methods in Immune Cell Atlas. d, UMAP plots of cell type labels and batches in Immune Cell Atlas using different integration methods. e, UMAP plots of cell type labels and batches using FADVI representations in batch subspace. f, Prioritized genes associated with cell type labels or batch effect in the whole dataset, individual cell types, and batches. High attribution variance suggested substantial expression variability across batches. Error bars represent standard error. g, Total integration scores across ablation settings in six benchmark datasets under different supervised settings with stratified label retention. We benchmarked FADVI with other integration methods on three tasks: atlas-scale scRNA-seq integration, single-cell ATAC sequencing (scATAC-seq) integration, and high-resolution ST with paired scRNA-seq integration. The top-performing methods from two consecutive benchmarking studies were selected for benchmarking, including semi-supervised scANVI 3 , and unsupervised scVI 7 , Combat 8 , Harmony 9 , and LIGER 10 . We also included another single-cell data disentanglement method DRVI 11 . For scRNA-seq integration, we evaluated FADVI on six datasets from a recent benchmarking study 1 : diabetic kidney disease dataset 12 , GTEX v9 13 , HypoMap 14 , Immune Cell Atlas 15 , Mouse Pancreatic Islet Atlas 16 , and Tabula Sapiens 17 . Across all datasets, both fully-supervised FADVI (with complete cell type labels) and semi-supervised FADVI (with partial labels) consistently outperformed competing methods, achieving substantially higher batch correction scores while maintaining comparable levels of biological variation conservation ( Fig. 1b ). Unsupervised FADVI had similar performance as other unsupervised methods. For instance, in Immune Cell Atlas with 329,762 cells from 12 batches, FADVI and semi-supervised FADVI improved the aggregated batch correction score by approximately 26% compared to other methods ( Fig. 1c ). Latent space visualization using Uniform Manifold Approximation and Projection (UMAP) further highlighted FADVI’s advantage, showing better separation of cell clusters alongside improved mixing of batches ( Fig. 1d ). In addition, when projecting into the batch-specific subspace, FADVI produced representations where batch-specific clusters were clearly distinguishable, confirming successful disentanglement of technical and biological variation. Similar patterns were observed across all scRNA-seq datasets (Supplementary Fig. 1-5). FADVI was also robust with a small subset of wrongly labeled cells (Supplementary Fig. 6). Moreover, the classification head in semi-supervised FADVI accurately predicted cell types for unlabeled cells, demonstrating its effectiveness in capturing biological variation even with incomplete label information (Supplementary Fig. 7). Leveraging FADVI’s ability to disentangle batch and biological variation, we applied Shapley Additive Explanations (SHAP) to identify genes associated with cell type labels and batch effects 18 . In Immune Cell Atlas, TRBC1 and TRAC emerged as the top genes linked to overall batch effects ( Fig. 1f ). At the cell type level, known marker genes were consistently prioritized for label prediction, while distinct sets of genes were associated with batch effects in different cell types, such as CTSD, LMNA in alveolar macrophage, and RGS1, HSPB1 in mast cells. We also detected batch-specific signatures, such as MTRNR2L12 in batches 582C/637C and GPR183 in D496 (Supplementary Fig. 8). We evaluated the contribution of individual loss components in FADVI under supervised and semi-supervised settings ( Fig. 1g , Supplementary Table 1). Removing label classification losses caused the largest decline, underscoring the essential role of supervised signal for representation learning. Adversarial terms were also beneficial, with complete removal leading to strong deterioration in batch correction. Among individual components, residual adversarial losses had minimal effect, while cross batch–label terms provided moderate benefits. The Cross-covariance loss, although not critical for predictive accuracy, contributed to more effective batch effect correction. Moreover, batch-level masking was generally more robust than stratified label retention for semi-supervised models. For scATAC-seq integration, we evaluated FADVI using mouse brain datasets from a previous benchmarking study 2 , including the full dataset (Large ATAC) and a downsampled version (Small ATAC). Each dataset was represented in three feature spaces: peaks, windows and gene activity. Both FADVI and semi-supervised FADVI outperformed competing methods, including LIGER, the top-ranked method in the original benchmarking ( Fig. 2a ) 2 . UMAP visualization further confirmed the disentangled variations across three feature spaces (Supplementary Fig. 9-14). Download figure Open in new tab Fig. 2. FADVI enables robust integration of scATAC-seq data, and high-resolution ST across diverse platforms with scRNA-seq data. a, FADVI shows higher integration scores with competing methods across two scATAC-seq datasets. b, FADVI shows higher integration scores for integrating ST with scRNA-seq data across three cancer types. FADVI(semi) and scANVI(semi): semi-supervised training with masked cell type labels in all ST data. c, Cell types labels of OV ST data overlaid on hematoxylin and eosin staining image. d, UMAP plots of cell type labels and batches in OV ST data using FADVI representations in batch subspace. e, UMAP plots of cell type labels and batches using different integration methods. The most challenging task was integrating high-resolution ST across diverse platforms with paired scRNA-seq data. We analyzed datasets profiling colon adenocarcinoma (COAD), hepatocellular carcinoma (HCC), and ovarian cancer (OV), generated using five ST platforms alongside scRNA-seq 19 . Among them, CoxMx, Stereo-seq, and Xenium provide single-cell resolution, whereas Visium HD captures 8 μm spots. In the semi-supervised setting, all ST cells or spots were treated as unlabeled, representing >99% of the data. For comparison, we also benchmarked Tangram 20 , a method designed for aligning ST with scRNA-seq data. Semi-supervised methods (FADVI and scANVI) outperformed unsupervised methods, while fully-supervised methods with complete cell type labels achieved best performance ( Fig. 2b ). Notably, FADVI surpassed scANVI in both semi- and fully-supervised contexts. Considering technologies as batches ( Fig. 2c ), FADVI effectively captured technology-associated variation ( Fig. 2d ). UMAP visualization showed the poorly batch-integrated latent representations across all methods except FADVI. However, semi-supervised FADVI also revealed limited integration between scRNA-seq and ST data, suggesting that scRNA-seq labels alone were insufficient to fully guide integration ( Fig. 2d , Supplementary Fig. 15-16). In summary, FADVI achieved robust integration across diverse scRNA-seq, scATAC-seq, and ST datasets, consistently outperforming state-of-the-art methods. By explicitly disentangling batch and biological variation, FADVI not only improves integration but also enables prioritization of features associated with cell type identity and batch effects. We anticipate that FADVI will provide a powerful framework for integrating large-scale single-cell and spatial omics datasets, particularly those affected by strong batch effects. Methods Model architecture FADVI explicitly separates latent representations into three parts: label-related factors ( z l ), b atch-related factors ( z b ), and residual factors ( z r ). Given an input observation x ∈ ℝ D , the encoder produces Gaussian parameters for each latent subspace: Each latent is then sampled using the reparameterization trick: The concatenated latent representation is decoded back into the input space: The VAE is trained with the evidence lower bound (ELBO): where p( z i )is the standard Gaussian prior, and β controls KL regularization strength. The first term corresponds to the reconstruction loss. To enforce semantic alignment, we attach fully-connected classification heads with supervised cross-entropy (CE) loss: To ensure each subspace is exclusive, fully-connected adversarial heads are trained via gradient reversal (GR) to prevent leakage: To further decorrelate the latent subspaces, we apply a cross-covariance penalty to minimizes statistical dependence between different latent factors: The total loss combines all components: For unlabeled cells,ℒ sup-l , ℒ adv-bl , ℒ adv-rl are set to 0. Batch integration benchmarking We benchmarked FADVI against several integration methods: scANVI 3 , scVI 7 , Combat 8 , Harmony 9 from Python package scib (v1.1.7) 2 ; LIGER from R package rliger (v2.2.1) 10 ; DRVI from Python package drvi-py (v0.1.9) 11 ; Tangram from Python package Tangram (v1.0.4) 20 . Running scDisco failed on all datasets so it is not included 4 . All methods except DRVI were run with default parameters. For scRNA-seq datasets, DRVI was run using additional parameters: {n_latent=128, max_epochs=400, early_stopping=False, n_epochs_kl_warmup=400}. For scRNA-seq and ST–scRNA-seq integration, the top 2,000 highly variable genes were used as input, while for scATAC-seq integration, all features (peaks, windows, or gene activity) were included. Latent representations generated by each method were evaluated using scib_metrics (v0.5.6) and visualized with UMAP using Scanpy (v1.11.1) 21 . FADVI and scANVI were tested under supervised, semi-supervised, and unsupervised settings. In the supervised setting, all cell type labels were provided as input, whereas in the unsupervised setting no labels were used. The semi-supervised setting simulated real-world scenarios by masking labels in specific subsets: half of the batches in scRNA-seq datasets, two smaller batches in scATAC-seq datasets, and all ST data in the paired ST-scRNA-seq datasets. For benchmarking, we used label subspace representations from supervised and semi-supervised FADVI, and the concatenated label and residual subspaces from unsupervised FADVI. Robustness to label noise Supervised methods can be sensitive to mislabeled data, which frequently occur in large-scale annotations. To assess FADVI’s robustness, we randomly shuffled 5% of the labels across all annotated cells in both fully and semi-supervised settings, simulating label noise in real datasets. We then evaluated performance using aggregated integration metrics and UMAP visualization. In both cases, FADVI produced results comparable to those obtained without label perturbation, indicating that its performance remains stable when a small subset of labels is wrong. Model prediction interpretability FADVI implements interpretability analysis using GradientShap from Captum to explain model predictions at the feature level 18 . It computes feature attributions by analyzing how each gene contributes to batch or label classification decisions through gradient-based methods. Batch predictions trace gradients through the batch subspace, while label predictions use the label subspace. Cell-level attributions are then transformed into feature-level importance. For each gene, FADVI computes mean attribution, standard deviation, and mean absolute attribution. High mean attribution or mean absolute attribution indicate features strongly associated with cell type labels or batch effect. Ablation study To evaluate the role of individual loss components, we conducted ablation experiments under supervised and semi-supervised regimes. Semi-supervised settings were generated using two strategies: (i) stratified label retention, where a subset of cells in each cell type were unlabeled, and (ii) batch-level masking, where entire batches were unlabeled. These strategies mimic annotation sparsity and heterogeneous batch structures commonly observed in single-cell studies. The network architecture was fixed across all experiments, and ablated models were created by selectively removing specific objectives: cross-covariance loss, label classification, batch classification, all adversarial terms, residual-space adversarial losses, or cross batch–label adversarial losses. Integration performance was assessed using biological conservation, batch correction, and total scores. Data availability Public datasets were available at: https://openproblems.bio/datasets (scRNA-seq), https://doi.org/10.6084/m9.figshare.12420968 (scATAC-seq), https://spatch.pku-genomics.org/#/download (ST with paired scRNA-seq). Code availability FADVI is available at https://github.com/liuwd15/fadvi . Author contributions WL conceived the project, developed FADVI method, and performed benchmarking. GQ performed ablation analysis. LMS and FJT made suggestions. All authors wrote and reviewed the manuscript. Competing interests F.J.T. consults for Immunai Inc., CytoReason Ltd, Cellarity, BioTuring Inc., and Genbio.AI Inc., and has an ownership interest in Dermagnostix GmbH and Cellarity. The remaining authors declare no competing interests. Acknowledgements This work was partially supported by National Institutes of Health grants U01AG079847 and R01LM012806. We thank the technical support of the Cancer Genomics Core supported by the Cancer Prevention & Research Institute of Texas (CPRIT) grant RP240610. WL was supported by John and Rebekah Harper Fellowship in Biomedical Sciences. WL and GQ were supported by CPRIT Fellowship in the Biomedical Informatics, Genomics and Translational Cancer Research Training Program (BIG-TCR, CPRIT RP210045). The funders had no role in the study design, data collection and analysis, decision to publish or prepare the manuscript. Funder Information Declared National Institutes of Health , U01AG079847 , R01LM012806 Cancer Prevention and Research Institute of Texas , RP240610 , RP210045 References 1. ↵ Luecken , M. D. et al. Defining and benchmarking open problems in single-cell analysis . Nat Biotechnol 43 , 1035 – 1040 , doi: 10.1038/s41587-025-02694-w ( 2025 ). OpenUrl CrossRef PubMed 2. ↵ Luecken , M. D. et al. Benchmarking atlas-level data integration in single-cell genomics . Nat Methods 19 , 41 – 50 , doi: 10.1038/s41592-021-01336-8 ( 2022 ). OpenUrl CrossRef PubMed 3. ↵ Xu , C. et al. Probabilistic harmonization and annotation of single-cell transcriptomics data with deep generative models . Mol Syst Biol 17 , e9620 , doi: 10.15252/msb.20209620 ( 2021 ). OpenUrl CrossRef PubMed 4. ↵ Liu , R. , Qian , K. , He , X. & Li , H. Integration of scRNA-seq data by disentangled representation learning with condition domain adaptation . BMC Bioinformatics 25 , 116 , doi: 10.1186/s12859-024-05706-9 ( 2024 ). OpenUrl CrossRef 5. Piran , Z. , Cohen , N. , Hoshen , Y. & Nitzan , M. Disentanglement of single-cell data with biolord . Nat Biotechnol 42 , 1678 – 1683 , doi: 10.1038/s41587-023-02079-x ( 2024 ). OpenUrl CrossRef PubMed 6. ↵ Gao , Y. , Dong , K. , Shan , C. , Li , D. & Liu , Q. Causal disentanglement for single-cell representations and controllable counterfactual generation . Nat Commun 16 , 6775 , doi: 10.1038/s41467-025-62008-1 ( 2025 ). OpenUrl CrossRef PubMed 7. ↵ Lopez , R. , Regier , J. , Cole , M. B. , Jordan , M. I. & Yosef , N. Deep generative modeling for single-cell transcriptomics . Nat Methods 15 , 1053 – 1058 , doi: 10.1038/s41592-018-0229-2 ( 2018 ). OpenUrl CrossRef PubMed 8. ↵ Johnson , W. E. , Li , C. & Rabinovic , A. Adjusting batch effects in microarray expression data using empirical Bayes methods . Biostatistics 8 , 118 – 127 , doi: 10.1093/biostatistics/kxj037 ( 2007 ). OpenUrl CrossRef PubMed Web of Science 9. ↵ Korsunsky , I. et al. Fast, sensitive and accurate integration of single-cell data with Harmony . Nat Methods 16 , 1289 – 1296 , doi: 10.1038/s41592-019-0619-0 ( 2019 ). OpenUrl CrossRef PubMed 10. ↵ Welch , J. D. et al. Single-Cell Multi-omic Integration Compares and Contrasts Features of Brain Cell Identity . Cell 177 , 1873 – 1887 e1817 , doi: 10.1016/j.cell.2019.05.006 ( 2019 ). OpenUrl CrossRef PubMed 11. ↵ Moinfar , A. A. & Theis , F. J. Unsupervised Deep Disentangled Representation of Single-Cell Omics . bioRxiv , 2024.2011.2006.622266 , doi: 10.1101/2024.11.06.622266 ( 2025 ). OpenUrl Abstract / FREE Full Text 12. ↵ Wilson , P. C. et al. Multimodal single cell sequencing implicates chromatin accessibility and genetic background in diabetic kidney disease progression . Nat Commun 13 , 5253 , doi: 10.1038/s41467-022-32972-z ( 2022 ). OpenUrl CrossRef PubMed 13. ↵ Eraslan , G. et al. Single-nucleus cross-tissue molecular reference maps toward understanding disease gene function . Science 376 , eabl4290 , doi: 10.1126/science.abl4290 ( 2022 ). OpenUrl CrossRef PubMed 14. ↵ Steuernagel , L. et al. HypoMap-a unified single-cell gene expression atlas of the murine hypothalamus . Nat Metab 4 , 1402 – 1419 , doi: 10.1038/s42255-022-00657-y ( 2022 ). OpenUrl CrossRef PubMed 15. ↵ Dominguez Conde , C. et al. Cross-tissue immune cell analysis reveals tissue-specific features in humans . Science 376 , eabl5197 , doi: 10.1126/science.abl5197 ( 2022 ). OpenUrl CrossRef PubMed 16. ↵ Hrovatin , K. et al. Delineating mouse beta-cell identity during lifetime and in diabetes with a single cell atlas . Nat Metab 5 , 1615 – 1637 , doi: 10.1038/s42255-023-00876-x ( 2023 ). OpenUrl CrossRef PubMed 17. ↵ Tabula Sapiens , C. et al. The Tabula Sapiens: A multiple-organ, single-cell transcriptomic atlas of humans . Science 376 , eabl4896 , doi: 10.1126/science.abl4896 ( 2022 ). OpenUrl CrossRef PubMed 18. ↵ Lundberg , S. M. & Lee , S.-I. A unified approach to interpreting model predictions . Advances in neural information processing systems 30 ( 2017 ). 19. ↵ Avraham-Davidi , I. et al. Spatially defined multicellular functional units in colorectal cancer revealed from single cell and spatial transcriptomics . bioRxiv , doi: 10.1101/2022.10.02.508492 ( 2025 ). OpenUrl Abstract / FREE Full Text 20. ↵ Biancalani , T. et al. Deep learning and alignment of spatially resolved single-cell transcriptomes with Tangram . Nat Methods 18 , 1352 – 1362 , doi: 10.1038/s41592-021-01264-7 ( 2021 ). OpenUrl CrossRef 21. ↵ Virshup , I. et al. The scverse project provides a computational ecosystem for single-cell omics data analysis . Nat Biotechnol 41 , 604 – 606 , doi: 10.1038/s41587-023-01733-8 ( 2023 ). OpenUrl CrossRef View the discussion thread. Back to top Previous Next Posted November 04, 2025. Download PDF Supplementary Material Email Thank you for your interest in spreading the word about bioRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following FADVI: disentangled representation learning for robust integration of single-cell and spatial omics data Message Subject (Your Name) has forwarded a page to you from bioRxiv Message Body (Your Name) thought you would like to see this page from the bioRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share FADVI: disentangled representation learning for robust integration of single-cell and spatial omics data Wendao Liu , Gang Qu , Lukas M. Simon , Fabian J. Theis , Zhongming Zhao bioRxiv 2025.11.03.683998; doi: https://doi.org/10.1101/2025.11.03.683998 Share This Article: Copy Citation Tools FADVI: disentangled representation learning for robust integration of single-cell and spatial omics data Wendao Liu , Gang Qu , Lukas M. Simon , Fabian J. Theis , Zhongming Zhao bioRxiv 2025.11.03.683998; doi: https://doi.org/10.1101/2025.11.03.683998 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Bioinformatics Subject Areas All Articles Animal Behavior and Cognition (7619) Biochemistry (17642) Bioengineering (13865) Bioinformatics (41863) Biophysics (21410) Cancer Biology (18548) Cell Biology (25437) Clinical Trials (138) Developmental Biology (13359) Ecology (19863) Epidemiology (2067) Evolutionary Biology (24288) Genetics (15587) Genomics (22467) Immunology (17704) Microbiology (40301) Molecular Biology (17142) Neuroscience (88447) Paleontology (666) Pathology (2825) Pharmacology and Toxicology (4815) Physiology (7634) Plant Biology (15109) Scientific Communication and Education (2042) Synthetic Biology (4285) Systems Biology (9812) Zoology (2268)
Text is read by the "Ask this paper" AI Q&A widget below.
Extraction quality varies by source — PMC NXML preserves structure
cleanly, OA-HTML may include some navigation residue, and OA-PDF can
have broken hyphenation. The publisher copy
(via DOI)
is the canonical version.