LN’s t-Test: A Principled Approach to t-Testing in Single-Cell RNA Sequencing
preprint
OA: closed
CC-BY-4.0
Abstract
Single-cell RNA sequencing (scRNA-seq) has revolutionized the study of cellular heterogeneity, yet differential gene expression (DGE) analysis remains hindered by inconsistencies in log fold change (LFC) estimation. Existing methods, such as those implemented in Scanpy and Seurat, rely on log-transformed count data with a pseudocount, introducing a bias that compromises the reliability of the statistical inference. In this work, we propose LN’s t -test, a novel approach to DGE testing that circumvents these biases by employing a log-Normal (LN) distribution-based LFC estimator. Our method jointly estimates the probability of non-zero expression and the mean of positive expression values, enabling an asymptotically unbiased and normally distributed LFC estimator with corresponding confidence intervals. Through extensive simulation studies, we demonstrate that LN’s t -test outperforms competing methods by reducing false discovery rates and providing more accurate effect size estimates. Notably, we leverage stochastic ordering theory to explain why conventional t -tests systematically mis-classify non-differentially expressed genes under realistic variance conditions. Our approach offers a theoretically principled and computationally efficient alternative for DGE analysis in scRNA-seq, with implications for improving the reliability and interpretability of single-cell transcriptomics studies. Code that implements the results is available on GitHub: https://github.com/okviman/DE-ZILN .
Full text
46,821 characters
· extracted from
preprint-html
· click to expand
LN’s t-Test: A Principled Approach to t-Testing in Single-Cell RNA Sequencing | bioRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-M677548'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search New Results LN’s t -Test: A Principled Approach to t -Testing in Single-Cell RNA Sequencing View ORCID Profile Oskar Kviman , View ORCID Profile Seong-Hwan Jun , Jens Lagergren doi: https://doi.org/10.1101/2025.03.12.642799 Oskar Kviman 1 KTH Royal Institute of Technology , Stockholm, Sweden 2 SciLifeLab , Solna, Sweden Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Oskar Kviman For correspondence: okviman{at}gmail.com okviman{at}kth.se Seong-Hwan Jun 3 University of Rochester , Rochester, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Seong-Hwan Jun Jens Lagergren 1 KTH Royal Institute of Technology , Stockholm, Sweden 2 SciLifeLab , Solna, Sweden Find this author on Google Scholar Find this author on PubMed Search for this author on this site Abstract Full Text Info/History Metrics Data/Code Preview PDF Abstract Single-cell RNA sequencing (scRNA-seq) has revolutionized the study of cellular heterogeneity, yet differential gene expression (DGE) analysis remains hindered by inconsistencies in log fold change (LFC) estimation. Existing methods, such as those implemented in Scanpy and Seurat, rely on log-transformed count data with a pseudocount, introducing a bias that compromises the reliability of the statistical inference. In this work, we propose LN’s t -test, a novel approach to DGE testing that circumvents these biases by employing a log-Normal (LN) distribution-based LFC estimator. Our method jointly estimates the probability of non-zero expression and the mean of positive expression values, enabling an asymptotically unbiased and normally distributed LFC estimator with corresponding confidence intervals. Through extensive simulation studies, we demonstrate that LN’s t -test outperforms competing methods by reducing false discovery rates and providing more accurate effect size estimates. Notably, we leverage stochastic ordering theory to explain why conventional t -tests systematically mis-classify non-differentially expressed genes under realistic variance conditions. Our approach offers a theoretically principled and computationally efficient alternative for DGE analysis in scRNA-seq, with implications for improving the reliability and interpretability of single-cell transcriptomics studies. Code that implements the results is available on GitHub: https://github.com/okviman/DE-ZILN . 1 Introduction Single-cell RNA sequencing (scRNA-seq) has emerged as a powerful technique for profiling gene expression at the resolution of individual cells, enabling detailed exploration of cellular heterogeneity and the discovery of novel biological insights [ 12 ]. Over the past decade, numerous approaches have been proposed for differential gene expression (DGE) analysis of scRNA-seq data, ranging from methods specifically tailored to the unique characteristics of single-cell assays [ 2 , 8 ] to well-established bulk RNA-seq tools such as edgeR [ 18 ] and DESeq2 [ 11 ], as well as classical tests like Welch’s t -test, a.k.a. the two-sample Student’s t -tests with unequal variance (we will from here refer to the test setup as “ t -test”, for simplicity) which is implemented in popular single-cell analysis packages such as Seurat [ 19 ] and Scanpy [ 26 ]. Despite this proliferation of methodologies, the majority of benchmarking efforts have focused on aspects such as statistical power, false discovery rates, and overall performance of the DGE tests [ 22 ]. Relatively little attention, however, has been given to the accuracy of one critical component of DGE analysis: log fold change (LFC) estimation. 4 The LFC is a key measure of effect size and is frequently used to interpret and prioritize biologically meaningful differences in gene expression. For instance, volcano plots, a popular visualization tool, display the fold change on the x -axis and the negative log p -values on the y -axis, highlighting genes with both a small p -value and a large absolute fold change. In typical scRNA-seq analysis pipelines, data are log-transformed or otherwise normalized as part of the recommended pre-processing workflow by Seurat and Scanpy. Furthermore, one of the common problems encountered in scRNA-seq data analysis is batch effect. Batch effect arises due to latent variation in genomic experiments, typically including experimental factors such as points, locations, and sequencer. Batch effect can lead to erroneous downstream analysis, including elevated false discovery rates [ 10 ]. Many batch correction methods require continuous expression values, making logarithmic transformation a necessary step [ 3 , 7 ]. However, in popular software packages such as Seurat and Scanpy, the average LFC is often computed directly from the sample means of these logarithmic transformed values, a procedure that can introduce bias due to zero inflation and other sources of technical noise [ 14 ]. An underappreciated consequence of this naive approach is that even when statistical tests are valid for hypothesis testing, the resulting confidence intervals and effect size estimates may be unreliable, ultimately limiting their utility for guiding follow-up experimental work or identifying robust biomarkers. In this work, we propose a new DGE test, LN’s t-test (pronounced Ellen’s t -test), which we derive from a novel LFC estimator that jointly estimates the probability of a gene expression count being non-zero and the expected value of positive gene expressions; notably, our approach circumvents the need to add pseudo-counts before log-transforming (a.k.a., log1p-transforming) the data. By constructing the two estimators as approximately log-Normal (LN) distributed, our LFC estimator is normally distributed with corresponding closed-form confidence intervals, permitting faithful application of the classical t -test framework for DGE testing in scRNA-seq. This is empirically manifested by evaluating LN’s t -test on simulated datasets that incorporate varying degrees of sparsity through adjustable negative binomial dispersion parameters. The simulated data experiments demonstrate that, in contrast to the tests implemented in Scanpy and Seurat, our method is robust against false discoveries and provides more accurate effect size estimates. Additionally, we utilize results from the stochastic ordering literature to connect the log1p transform to systematically falsely discovering genes as differentially expressed. Our contributions can be chronologically summarized as follows: In Sec. 3, we analyze the LFC estimation and the t -test implementations in Seurat and Scanpy, high-lighting their limitations. Using the theory of statistical ordering, we reason that the current t -tests can systematically result in false discoveries. The claim is verified in multiple experiments in Sec. 5. In Sec. 4, we construct a novel LFC estimator and derive our LN’s t -test. In Sec. 5, we demonstrate the superiority and reliability of LN’s t -test compared to tests provided by Scanpy and Seurat. In highlighting the importance of robust LFC estimation, we aim to spur a broader discussion on refining the methodological underpinnings of DGE analysis in single-cell genomics, thereby enhancing the reliability of biological insights drawn from scRNA-seq studies. 2 Background In scRNA-seq, y is an n × g gene expression matrix, with n the number of cells and g the number of genes, so that y ij ∈ [0, ∞) is a (UMI or read) count of gene expressions from cell i ∈ [ 1 , n ] and gene j ∈ [ 1 , g ]. In a differential gene expression experiment, one additionally obtains x ∈ [0, ∞) n′×g corresponding to an alternative condition (examples of comparative conditions include treatment vs. control, or different cell types), and then asks whether there is a significant LFC between the two conditions with respect to a specific gene j . Mathematically, the LFC measures the difference in the log 2 -transformed expected expression values for gene j across conditions, where Y j , X j are the random variables that, respectively, follow the condition-specific distributions of the j -th gene’s expression values. The expected values are unknown, so they must be approximated using estimators. To decide whether an LFC estimate is significantly different from zero, scRNA-seq practitioners employ popular and well-known statistical frameworks, like the t -test. In the t -test, one constructs a t statistic which compares the difference between two inferred means and scaled by the standard error of the estimates , that is For a critical value, t α,ν , where α is a given significance level and ν the degrees of freedom, confidence intervals (CIs) that quantify the uncertainty in the inferred difference in means can be obtained following [ 24 ] While this test appears appropriate for LFC estimation, it presents a significant issue, which we refer to as issue A : In the numerator in Eq. (2) there is a difference in means, while in Eq. (1) there is a difference in log-transformed means. In the next section we demonstrate how two popular scRNA-seq packages estimate LFCs and handle the mentioned issue that hinders straightforward application of the t -test. 3 Differential Gene Expression Testing with Scanpy and Seurat Here we describe how Scanpy [ 26 ] and Seurat [ 1 , 4 , 5 , 19 , 23 ] implement LFC estimation and t -testing, and demonstrate that their LFC estimates are not necessarily in the CIs used in their t -tests. Moreover, using results from the statistical ordering literature, we cast light on a crucial issue ( C ) in the t -test implementations resulting in an increased risk of classifying genes as DEGs. 3.1 LFC Estimators In a recent work [ 16 ], a batch-effect adjusted version 5 of the LFC estimator in Scanpy was given as with ϵ = 10 −9 . Notably, ϵ is added to avoid log 2 (0) but its effect does not diminish with the number of cells. Meanwhile, since the publication of [ 16 ], Seurat’s LFC estimator has been updated in their software, where are log1p-transformed such that 1 and similarly define x ij , and so we can simply express S 2 as Seurat also allows to specify a value ϵ for pseudocount rather than 1, where by default ϵ = 10 −9 as in Eq. (4) . In S 2 and S 3 , the effect of ϵ diminishes in the limit of n and n ′ . Importantly, although these are the LFC estimates that the softwares return, 6 these are not the estimated effect sizes (the numerator in Eq. (2) ) used to construct t statistics in the softwares. This causes a new issue ( B ), discussed in the next section. 3.2 Approaches to t -Testing As mentioned in Sec. 2, the t -test framework is an interesting test to use to measure the significance of LFC estimates. However, it is not trivial to construct a t statistic in LFC estimation in scRNA-seq (issue A ). For instance, there is no clear way to obtain standard errors for S 1 , S 2 or S 3 , and, moreover, an estimator of the LFC based on the sample mean of the log-transformed samples is not defined since a gene can be unexpressed (and so its log-transform is undefined). Scanpy and Seurat handle this by using the log1p transform, such that, matching terms to those in Eq. (2) we denote , and so on for and the approximate variances, . Thus, the estimated LFC of gene j used in the t -tests in both softwares is ,not S 1 , S 2 or S 3 . This discrepancy causes issue B , while the log1p transformation of the data gives rise to a third, critical issues ( C ). Issues B and C are described in the subsequent two paragraphs. B Scanpy and Seurat’s LFC estimates are not necessarily in the inferred confidence intervals Notably, the difference in sample means of the log1p-transformed counts, 7 , is not the same as those in Eq. (4) or (7) . Hence, the LFCs estimated by Scanpy and Seurat are not necessarily found within the CIs that their own softwares produce. The provided t -tests are therefore significance tests of the difference in log1p-transformed data, and not of S 1 , S 2 or S 3 . In Fig. 1 we show that S 1 and S 3 can indeed be found outside their corresponding CIs. Download figure Open in new tab Fig. 1: (a) The LFC estimates produced by Scanpy and Seurat ( S 1 from Eq. (4) and S 3 from (7), respectively) are not necessarily within the softwares’ corresponding t -test CIs (see Eq. (3) ). Here (details in Sec. 5.2), Scanpy’s LFC estimates (blue) are above the inferred CIs 150 times out of 1500, while Seurat’s estimates (black) are never in the CIs. (b) LN’s t -test returns accurate LFC estimates that, by design, are centered in the corresponding CIs. C False discoveries will occur under mild assumptions The stochastic ordering literature states that if X is a mean-preserving contraction of Y , i.e. X is greater than Y in concave order, then Var( X ) ≤Var( Y ), 𝔼[ X ] = 𝔼[ Y ] and 𝔼[log( X + 1)] ≥ 𝔼[log( Y + 1)] (a full treatment of concave orders is given in [ 20 , Chapter 3.A]). Fascinatingly, this result has negative implications on the t -tests in Scanpy and Seurat. Concretely, commonly occurring pairs of gene-expression distributions (like NB vs. NB or NB vs. Poisson) are concavely ordered if they share mean values but have different variances. Therefore, if we are testing genes that are not DEGs (𝔼[ X ] = 𝔼[ Y ]) and assume that their expression distributions are, for example, NBs with Var( X ) 𝔼[log( Y + 1)] (note that the inequalities are strict). Hence, on average or as the numbers of cells grow, the Scanpy and Seurat t -tests will falsely discover the genes as DEG. Assuming that data from non-DEGs follow NBs and/or Poissons with dissimilar variances are clearly mild assumptions as the mentioned distributions are common distributional assumptions in scRNA-seq [ 21 , 25 ], and since non-DEGs can reasonably have differing variances across conditions (due to difference in cell types, stressed cells, environmental changes or treatment responses, for example). In fact, the abundance of non-DEGs with unequal variance is one of the main motivations for doing non-LFC-based DGE testing [ 6 , 9 ]. To identify concavely ordered distributions we can use Theorem 3.A.44 in [ 20 ]. Namely, if we compare the probability density or mass functions of X and Y, p X ( u ) and p Y − ( u ), and if the sign of p Y ( u ) − p X ( u ) changes two times (from positive to negative and then back to positive) as u grows, then X is a mean-preserving contraction of Y . See Fig. 2 for an illustration. In our experiments we show how discrepancies in variance between NB distributions can result in 100% false positive rates. Download figure Open in new tab Fig. 2: A typical example of two gene expression distributions, p X and p Y , which have the same mean, 𝔼[ X ] = 𝔼[ Y ] = µ , but different variances (small difference in dispersion parameters). This is also an example of when X is a mean-preserved contraction of Y , which implies that 𝔼[log( X + 1)] ≥ 𝔼[log( Y + 1)]. Hence, Scanpy and Seurat t -tests will falsely discover the corresponding genes as DEGs. 4 Proposing LN’s t -Test Here we propose our asymptotically unbiased LFC estimator and derive its associated t statistic and CIs. For clarity, we drop the gene index. To construct our estimator, we start by noting that the expected gene expression can be decomposed as Our estimator separately estimates the two terms in Eq. (9) , i.e. the probability of an observation being positive, P ( Y > 0), and the expected positive gene expression, 𝔼[ Y | Y > 0]. Our objective is to find an estimator of 𝔼[ Y ] that results in an asymptotically unbiased and normally distributed estimator of Eq. (1) . To do this, we estimate P ( Y > 0) and 𝔼[ Y | Y > 0] using two log-normally distributed estimators, leveraging the following lemma. Lemma 1. If and are independent, then and Proof . See Appendix A. From the lemma it follows that, if U, V, U ′ , V ′ are independent log-normally distributed r.v.s, then log 2 ( UV ) − log 2 ( U ′ V ′ ) is normally distributed, a property we will use to estimate the LFC ( Eq. (1) ). 4.1 The Estimator of P ( Y > 0) In this section we develop a novel approximation scheme of the Beta probability density function (pdf) by carefully parameterizing a log-normal pdf. We emphasizing the novelty of the estimator proposed here, and that, as far as we are aware, there are no previous works on Beta pdf approximation using an LN pdf, especially using our parameterization. We conclude the section by determining that is an asymptotically unbiased estimator of P ( Y > 0) via the Theorem 1. A standard assumption in e.g. Bayesian hypothesis testing (see Sec. 3.10.2.4 in [ 15 ]) on an estimator of P ( Y > 0) is that it is Beta distributed, so we make the standard assumption that However, in order to apply Lemma 1, we are requiring a log-normally distributed estimator. To obtain this, we note that log θ Y is approximately normally distributed, especially when the Beta distribution has non-negative skewness (i.e., when a Y ≤ b Y ; see Fig. 5 for an illustrative example). Conveniently, there are known expressions for the mean and the variance of a log-transformed Beta random variable. Define ψ 0 ( z ) to be a second-order approximation of the digamma function and ψ 1 ( z ) to be a first-order approximation of the trigamma function 8 then Given data y 1: n let and using δ 0) = 0. Note that if P ( Y > 0) < 1, then . Recalling that the mean of an LN distribution with parameters µ and σ 2 is ,we use equations (13-14) to parameterize an LN distribution, we let and so, by approximating a Y and b Y with â Y and ,our estimator of P ( Y > 0) is (see the proof of Theorem 1 for a derivation) Next, we determine that is an asymptotically unbiased estimator of P ( Y > 0) via the following theorem. To ensure that Eq. (1) is defined, we will assume that P ( Y > 0) > 0. Theorem 1. Assuming that P ( Y > 0) > 0, the log-normally distributed estimator is an asymptotically unbiased estimator of P ( Y > 0), with a non-asymptotic bias that is strictly upper bounded by δ/n . Proof . See Appendix A. As we will see, our approximation scheme proposed here conveniently allows us to use Lemma 1 to construct an asymptotically unbiased LFC estimator with closed-form CIs. 4.2 The Estimator of 𝔼[ Y | Y > 0] In accordance with Eq. (15) , we refer to the number of samples with positive counts as n + and term the index set of these samples as .Here we have filtered out zero-counts, so we can log-transform the sample mean of the remaining, positive samples without adding any pseudocount. This sample mean is consequently strictly positive, and so, motivated by models like the Poisson LN in scRNA-seq (see, e.g., [ 21 ] where the mean of the Poisson expression distribution is log-normally distributed), and the log-Normal Cox process [ 13 ], we let We emphasize that modeling as log-normally distributed does not imply an LN assumption on the individual samples, y i . In fact, a sum of LN random variables, scaled or otherwise, is not LN in general and so Eq. (19) implies that the positive gene expressions are not log-normally distributed. We parameterize the LN sample mean distribution by solving the following system of equations resulting in Substituting the unknown terms with our approximations, i.e. and where we conclude that our resulting estimator of the mean of the positive gene expressions, ,follows an LN distribution with estimated parameters and Our estimator is clearly unbiased since . Algorithm 1 LN’s t -test (gene wise) Download figure Open in new tab Algorithm 2 Get Estimators Download figure Open in new tab 4.3 LN’s t -Test and Our Proposed LFC Estimator In the above sections, we have constructed two LN estimators of P ( Y > 0) and 𝔼[ Y | Y > 0], respectively. We can analogously obtain the estimators of the equivalent quantities w.r.t. X . As such, we now use Lemma 1 to conclude that is LN, so is also LN, why, consequently, our LFC estimator, is normally distributed with squared standard errors (add Eq. (14) with Eq. (26) ) As described in Sec. 2, dividing the estimated effect size, S LN , by the standard error, ,yields a t statistic, ,and its corresponding closed-form CIs centered around the LFC estimate Moreover, S LN is asymptotically unbiased, as stated in the following theorem. Theorem 2. As n and n ′ tend to infinity, S LN is an unbiased estimator of the LFC (Eq . (1) ). Proof . See Appendix A. We provide an algorithmic description in Algorithm 1. Algorithm 1 and 2 consist of series of 𝒪 (1), 𝒪 ( n ), 𝒪 ( n + ), 𝒪 ( n ′ ) and operations. Since 𝒪 ( n ) ≥ 𝒪 ( n + ) and ,summing the costs of these operations results in an 𝒪 ( n + n ′ ) complexity. Algorithm 1 is called for all g genes so the total complexity of LN’s t -test is 𝒪 ( gn + gn ′ ), which should be the same for the t -tests in Scanpy and Seurat. 5 Experiments The scope of our work is focused on fast, simple and popular DGE methods at single cell level, so we compare LN’s t -test with the t -test (with and without overestimated variance [ 26 ]) and the Wilcoxon rank-sum test (with and without the limma package [ 17 ]). All methods used CP10K normalization and Bonferroni 9 corrected p values and a 0.05 significance level. The normalized data was log1p-transformed before inputted to the Scanpy ( rank_genes_groups , following [ 16 ]) and Seurat ( FindMarkers ) DE methods. Here we generate read counts from two negative binomial (NB) distributions, one per condition, with the mean-dispersion parameterization, i.e., ,such that and .To simulate batch effects, we multiply the means of half of the cells in each condition by e λ . We experiment with multiple different combinations of parameters, each setup is described in detail below. For each setup, we then test the DGE capabilities of the different methods and their LFC estimation accuracies. To do so, we measure DGE classification accuracy, precision, true/false positive/negative rates (TPR/FPR/TNR/FNR) and F1 scores, and RMSE scores for the LFCs. All LFC estimators used to report the RMSE scores were implemented in python . We let n = n ′ = 1000, and g = 1500. Furthermore, Scanpy’s Wilcoxon and t -test (and the overestimated variance t -test) implementations produced (close to or) identical results to Seurat’s Wilcoxon (and Wilcoxon limma) and t -tests. As such we refer to these methods simply as t -test or Wilcoxon in the tables here. In Appendix B we include box plots demonstrating the results of all software-specific methods, too. First, we demonstrate in Fig. 3 how Scanpy and Seurat’s t -tests cannot recognize non-DGEs when two NBs are concavely ordered (seen as we vary the dispersion of the distribution of Y ; see C in Sec. 3 for an explanation of concave orders). The Scanpy t -test results apply to Seurat, too, as their implementations are the same. Meanwhile, LN’s t -test is robust to these distributional discrepencies in terms of accuracy and FPR. Download figure Open in new tab Fig. 3: Scanpy and Seurat are highly sensitive to discrepancies between Var( X ) and Var( Y ). FPR is evaluated against different combinations of variances. There are no DEGs in this experiment, and the means of the NBs are 10 and 100 in the left and right columns, respectively. Scanpy’s results degenerate even for small discrepancies in condition variances, reporting 100% false positives as Var( X ) and Var( Y ) diverge. LN’s t -test, on the other hand, correctly classifies all genes as non-DEGs. 5.1 Densely Expressed Gene Data We fix , and let for all j that are non-DEGs (on average 90% of the genes), and let where h j ∼ 𝒩 (15, 25) for all DEGs. We then vary in the set {0.1, 0.2, 1.}. When we do not simulate batch effects ( λ = 0), we obtain the results (averaged over 20 runs) in Table 1 and Fig. 8 , and some corresponding volcano plots in Fig. 6 - 7 . When λ = 1 and so ) we obtain the results in Table 4 and Fig. 1 , where we, in the latter, can visually compare the quality of the LFC estimates between S 1 , S 3 and S LN , while empirically confirming B ; S 1 and S 3 are not necessarily in the CIs outputted from their corresponding softwares. In Fig. 4(a) the RMSE results for the different estimators are reported. View this table: View inline View popup Download powerpoint Table 1: ( λ = 0 and Densely Expressed Gene Data ) Comparison of scores between our (underbarred) LN’s t -tests and the methods implemented in Scanpy and Seurat. The results are averaged over 20 runs and the data was generated from two NBs for each gene. See Sec. 5.1 for more details. Notably, as ϕ X diverges from ϕ Y the performances of all methods degenerate, except for ours. Download figure Open in new tab Fig. 4: LFC estimation results (averaged over 20 runs) for the different NB simulated data settings from Sec. 5.1 and 5.2, but with a more exhaustive list of as indicated by the ticks on the x -axes. Dashed lines represent results when the data does not have batch effects ( λ = 0). S LN and S 3 give identical results. t -test denotes the effect size estimate in Seurat and Scanpy, i.e., the difference in sample means of log1p-transformed CP10K normalized counts (see B for details). Download figure Open in new tab Fig. 5: Visualization of the similarities of a Beta distribution and its log-Normal approximation. The bottom axes are aligned in each column. Top left: Samples drawn from a Beta( a = 5, b = 500). Top right: The log-transformed samples from the Beta distribution and the estimated mean of the distribution (dashed blue). Bottom left: Samples drawn from log 𝒩 ( ψ 0 ( a ) − ψ 0 ( a + b ), ψ 1 ( a ) − ψ 1 ( a + b ). Bottom right: Samples drawn from 𝒩 ( ψ 0 ( a ) − ψ 0 ( a + b ), ψ 1 ( a ) − ψ 1 ( a + b ). Download figure Open in new tab Fig. 6: Gene counts from non-DEGs were simulated using the NB parameters and and and .Every 10th gene, j ′ , is differentially expressed with where .Here λ = 0, but the results do not differ much when λ = 1 for this particular combination of NB parameters. Download figure Open in new tab Fig. 7: Gene counts from non-DEGs were simulated using the NB parameters and ,and and . Every 10th gene, j ′ , is differentially expressed with where . Here λ = 0, again, the results are only slightly worse when λ = 1 for this particular combination of NB parameters, too. Download figure Open in new tab Fig. 8: ( λ = 0 and Densely Expressed Data ) Box plot results produced by the methods on the NB simulated data, complementing the results given in Table 1 . Download figure Open in new tab Fig. 9: ( λ = 1 and Densely Expressed Data ) Box plot results produced by the methods on the NB simulated data, complementing the results given in Table 4 . 5.2 Sparsely Expressed Gene Data Here non-DEGs instead have mean while and .10% of all genes are DEGs with random up-regulation given by an additive noise term | h j |where h j ∼ 𝒩 (0, 25). When λ = 0, we obtain the results (averaged over 20 runs) in Table 2 and in Fig. 10 , while for λ = 1 we report the results in Table 3 and in Fig. 11 . In Fig. 4(b) the RMSE results for the different estimators are reported. View this table: View inline View popup Download powerpoint Table 2: ( λ = 0 and Sparsely Expressed Gene Data ) Comparison of scores between our (underbarred) LN’s t -tests and the methods implemented in Scanpy and Seurat. The results are averaged over 20 runs and the data was generated from two NBs for each gene. See Sec. 5.2 for more details. View this table: View inline View popup Download powerpoint Table 3: ( λ = 1 and Sparsely Expressed Gene Data ) Comparison of scores between our (underbarred) LN’s t -tests and the methods implemented in Scanpy and Seurat. The results are averaged over 20 runs and the data was generated from two NBs for each gene. See Sec. 5.2 for more details. Notably, as ϕ X diverges from ϕ Y the performances of all methods degenerate, except for ours. View this table: View inline View popup Download powerpoint Table 4: ( λ = 1 and Densely Expressed Gene Data ) Comparison of scores between our (underbarred) LN’s t -tests and the methods implemented in Scanpy and Seurat. The results are averaged over 20 runs and the data was generated from two NBs for each gene. See Sec. 5.1 for more details. Download figure Open in new tab Fig. 10: ( λ = 0 and Sparsely Expressed Data ) Box plot results produced by the methods on the NB simulated data, complementing the results given in Table 2 . Download figure Open in new tab Fig. 11: ( λ = 1 and Sparsely Expressed Data ) Box plot results produced by the methods on the NB simulated data (with batch effect), complementing the results given in Table 3 . 6 Discussion One interesting observation from the LFC estimates in Fig. 4 is that, due to their asymptotic properties, the biases of S LN and S 3 both seem to vanish for the choices of sample sizes used (the RMSEs are not zero due to the CP10K normalization, without normalization and no batch effect the RMSEs for S LN and S 3 would be zero). Importantly, however, S 3 does not have known confidence intervals, in contrast to S LN (see Eq. (30) ). Another observation is that S 1 , surprisingly, performs worse than the difference in sample means of the log-transformed counts (the effect size estimate in the t -test). The poor RMSE scores from the t -test (green line) are expected from the stochastic ordering theory (see Sec. 3.2) and aligned with the degenerating DGE test performances we observed from the methods implemented by Scanpy and Seurat. The DGE results in tables 1 - 4 and the corresponding box plots 8-11, the volcano plots in Fig. 6 - 7 as well as the results in Fig. 3 collectively present a clear demonstration of the systematic false discoveries induced by the log1p transform, as we predicted via the theory of stochastic orderings in Sec. 3. Namely, Scanpy and Seurat estimate the effect size in their t statistic based on the difference of sample means of log1p transformed data which is a biased estimator of the LFC in Eq. (1) under mild assumptions in the scRNA-seq setting. Apparently also the Wilcoxon test (implemented in Scanpy, Seurat and limma) struggles in a similar manner in all our experiments, promoting further investigation of issue C ‘s implications on other tests. On the other hand, our LN’s t -test performs consistently well in all considered experimental setups. 7 Conclusion We have proposed LN’s t -test, a principled approach to t -testing in scRNA-seq that does not suffer from the biased LFC estimation procedure used in the t -tests in Scanpy and Seurat. Using the theory of stochastic ordering, we predicted that Scanpy and Seurat’s t -tests would produce high FPRs under mild assumptions, and verified our claim in extensive simulation studies. We further critically investigated the two softwares by comparing their reported LFC estimates and the ones used in the t -test, finding that their reported LFCs are not necessarily covered by the CIs used in the DGE tests. Future work consists of testing our connection between concavely ordered gene expression distributions and FPRs on interesting real data cases, while simultaneously applying LN’s t -test. Moreover, we look forward to exploring new LFC estimators within the LN’s t -test framework and identifying the failure modes of our proposed DGE testing framework. A Proofs In this section we have collected the proofs to the theorems and lemmas in the main text. We start with Lemma 1. Lemma 1. If and are independent, then and Proof . The log of UV is a sum of two normally distributed r.v.s, and so This extends trivially to U/V yielding We now continue with the proof of the asymptotic unbiasedness of the P ( Y > 0) estimator from Theorem 1. Before doing so, we leave two notable comments on the result: Although it is necessary to assume P ( Y > 0) > 0 for the LFC to be defined, the estimator would still be asymptotically unbiased if P ( Y > 0) = 0; the upper-bound on the bias would not be strict, but would go to zero at the same rate. Second, for most values of P ( Y > 0) and n (the number of cells), the bias is practically zero. Theorem 1. Assuming that P ( Y > 0) > 0, the log-normally distributed estimator is an asymptotically unbiased estimator of P ( Y > 0), with a bias that is strictly upper bounded by 1 /n . Proof . We begin by taking the expectation of ,where and so where the first term can be rewritten using a standard probability theoretic identity as while the second term, i.e. the bias, can be neatly quantified with some additional steps. First, recall that δ 0) > 0, and note that n + ∼ Binomial( n, P ( Y > 0)). Then, by first noting that we can use the law of the unconscious statistician which, via the Binomial theorem and by substituting 1 − P ( Y > 0) = P ( Y = 0), can be rewritten as Since δ 0) > 0 ⇒ P ( Y = 0) < 1, then 𝔼[ δ n + ] 0), with a strict upper-bound that is only achieved in the uninteresting case where P ( Y = 0) = 1, and that asymptotically goes to zero with n . For completeness, with Theorem 2. As n and n ′ tend to infinity, S LN is an unbiased estimator of the LFC (Eq . (1) ). Proof . We want to show that Recall that via the law of large numbers (LLN), we have that for a sequence of i.i.d. draws z 1 , …, z n with mean 𝔼[ Z ] and a function f that is continuous in 𝔼[ Z ], .This result thus applies to S LN , since 𝔼[ Y ] and 𝔼[ X ] are assumed to be positive and where f = log 2 . Hence, using ( i ) the result from Theorem 1, namely that and are asymptotically unbiased estimators of P ( Y > 0) and P ( X > 0), respectively, and ( ii ) that and are merely sample-mean estimators of 𝔼[ Y | Y > 0] and 𝔼[ X | X > 0] (hence they are unbiased, also via LLN), we have where we know that for non-negative r.v.s, 𝔼[ Z ] = P ( Z > 0)𝔼[ Z | Z > 0], and so B Additional Results Here we mainly include plots that are referenced from the main text. Footnotes https://github.com/okviman/DE-ZILN ↵ 4 DGE analysis can also be performed by comparing the variance of gene expression distributions instead of the LFC [ 6 ], but the topic of this work is LFC-based DGE analysis. ↵ 5 To properly analyze the estimators and due to the abundance of various normalization schemes, we study the estimators independently of normalization. In our experiments, however, we report our results using count per 10,000 (CP10K). ↵ 6 E.g. when running adata.uns[“rank_genes_groups”][“logfoldchanges”] after sc.tl.rank_genes_groups in Scanpy. 7 The difference is scaled by 1 / log(2) to make it comparable with the LFC, which is computed with base-2. ↵ 8 The second- and first-order approximations of the digamma and trigamma functions ensure that the estimator in Eq. (17) is unbiased, while higher order approximations did not notably improve the results in our experiments. ↵ 9 The default in Scanpy is Benjamini-Hochberg, but this correction resulted in even worse FPRs as it is less conservative. References 1. ↵ Butler , A. , Hoffman , P. , Smibert , P. , Papalexi , E. , Satija , R. : Integrating single-cell transcriptomic data across different conditions, technologies, and species . Nature biotechnology 36 ( 5 ), 411 – 420 ( 2018 ) OpenUrl CrossRef PubMed 2. ↵ Finak , G. , McDavid , A. , Yajima , M. , Deng , J. , Gersuk , V. , Shalek , A.K. , Slichter , C.K. , Miller , H.W. , McElrath , M.J. , Prlic , M. , Linsley , P.S. , Gottardo , R. : MAST: a flexible statistical framework for assessing transcriptional changes and characterizing heterogeneity in single-cell RNA sequencing data . Genome Biol . 16 , 278 (Dec 2015 ) OpenUrl CrossRef PubMed 3. ↵ Haghverdi , L. , Lun , A.T.L. , Morgan , M.D. , Marioni , J.C. : Batch effects in single-cell RNA-sequencing data are corrected by matching mutual nearest neighbors . Nat. Biotechnol . 36 ( 5 ), 421 – 427 (Jun 2018 ) OpenUrl CrossRef PubMed 4. ↵ Hao , Y. , Hao , S. , Andersen-Nissen , E. , Mauck , W.M. , Zheng , S. , Butler , A. , Lee , M.J. , Wilk , A.J. , Darby , C. , Zager , M. , et al : Integrated analysis of multimodal single-cell data . Cell 184 ( 13 ), 3573 – 3587 ( 2021 ) OpenUrl CrossRef PubMed 5. ↵ Hao , Y. , Stuart , T. , Kowalski , M.H. , Choudhary , S. , Hoffman , P. , Hartman , A. , Srivastava , A. , Molla , G. , Madad , S. , Fernandez-Granda , C. , et al : Dictionary learning for integrative, multimodal and scalable single-cell analysis . Nature biotechnology 42 ( 2 ), 293 – 304 ( 2024 ) OpenUrl CrossRef PubMed 6. ↵ Ho , J.W. , Stefani , M. , Dos Remedios , C.G., Charleston , M.A. : Differential variability analysis of gene expression and its application to human diseases . Bioinformatics 24 ( 13 ), i390 – i398 ( 2008 ) OpenUrl CrossRef PubMed Web of Science 7. ↵ Korsunsky , I. , Millard , N. , Fan , J. , Slowikowski , K. , Zhang , F. , Wei , K. , Baglaenko , Y. , Brenner , M. , Loh , P.R. , Raychaudhuri , S. : Fast, sensitive and accurate integration of single-cell data with harmony . Nat. Methods 16 ( 12 ), 1289 – 1296 (Dec 2019 ) OpenUrl CrossRef PubMed 8. ↵ Korthauer , K.D. , Chu , L.F. , Newton , M.A. , Li , Y. , Thomson , J. , Stewart , R. , Kendziorski , C. : A statistical approach for identifying differential distributions in single-cell RNA-seq experiments . Genome Biol . 17 ( 1 ), 222 (Oct 2016 ) OpenUrl CrossRef PubMed 9. ↵ Korthauer , K.D. , Chu , L.F. , Newton , M.A. , Li , Y. , Thomson , J. , Stewart , R. , Kendziorski , C. : A statistical approach for identifying differential distributions in single-cell rna-seq experiments . Genome biology 17 , 1 – 15 ( 2016 ) OpenUrl CrossRef PubMed 10. ↵ Leek , J.T. , Scharpf , R.B. , Bravo , H.C. , Simcha , D. , Langmead , B. , Johnson , W.E. , Geman , D. , Baggerly , K. , Irizarry , R.A. : Tackling the widespread and critical impact of batch effects in high-throughput data . Nat. Rev. Genet . 11 ( 10 ), 733 – 739 (Oct 2010 ) OpenUrl CrossRef PubMed Web of Science 11. ↵ Love , M.I. , Huber , W. , Anders , S. : Moderated estimation of fold change and dispersion for RNA-seq data with DESeq2 . Genome Biol . 15 ( 12 ), 550 ( 2014 ) OpenUrl CrossRef PubMed 12. ↵ Macosko , E.Z. , Basu , A. , Satija , R. , Nemesh , J. , Shekhar , K. , Goldman , M. , Tirosh , I. , Bialas , A.R. , Kamitaki , N. , Martersteck , E.M. , Trombetta , J.J. , Weitz , D.A. , Sanes , J.R. , Shalek , A.K. , Regev , A. , McCarroll , S.A. : Highly parallel genome-wide expression profiling of individual cells using nanoliter droplets . Cell 161 ( 5 ), 1202 – 1214 (May 2015 ) OpenUrl CrossRef PubMed 13. ↵ Møller , J. , Syversveen , A.R. , Waagepetersen , R.P. : Log gaussian cox processes . Scandinavian journal of statistics 25 ( 3 ), 451 – 482 ( 1998 ) OpenUrl CrossRef Web of Science 14. ↵ Moses , L. , Einarsson , P.H. , Jackson , K. , Luebbert , L. , Booeshaghi , A.S. , Antonsson , S. , Bray , N. , Melsted , P. , Pachter , L. : Voyager: exploratory single-cell genomics data analysis with geospatial statistics . bioRxivorg (Aug 2023 ) 15. ↵ Murphy , K.P. : Probabilistic machine learning: Advanced topics . MIT press ( 2023 ) 16. ↵ Pullin , J.M. , McCarthy , D.J. : A comparison of marker gene selection methods for single-cell rna sequencing data . Genome Biology 25 ( 1 ), 56 ( 2024 ) OpenUrl CrossRef PubMed 17. ↵ Ritchie , M.E. , Phipson , B. , Wu , D. , Hu , Y. , Law , C.W. , Shi , W. , Smyth , G.K. : limma powers differential expression analyses for rna-sequencing and microarray studies . Nucleic acids research 43 ( 7 ), e47 – e47 ( 2015 ) OpenUrl CrossRef PubMed 18. ↵ Robinson , M.D. , McCarthy , D.J. , Smyth , G.K. : edgeR: a bioconductor package for differential expression analysis of digital gene expression data . Bioinformatics 26 ( 1 ), 139 – 140 (Jan 2010 ) OpenUrl CrossRef PubMed Web of Science 19. ↵ Satija , R. , Farrell , J.A. , Gennert , D. , Schier , A.F. , Regev , A. : Spatial reconstruction of single-cell gene expression data . Nature biotechnology 33 ( 5 ), 495 – 502 ( 2015 ) OpenUrl CrossRef PubMed 20. ↵ Shaked , M. : Stochastic orders . Springer Science & Business Media ( 2007 ) 21. ↵ Silverman , J.D. , Roche , K. , Mukherjee , S. , David , L.A. : Naught all zeros in sequence count data are the same . Computational and structural biotechnology journal 18 , 2789 – 2798 ( 2020 ) OpenUrl CrossRef 22. ↵ Soneson , C. , Robinson , M.D. : Bias, robustness and scalability in single-cell differential expression analysis . Nat. Methods 15 ( 4 ), 255 – 261 (Apr 2018 ) OpenUrl CrossRef PubMed 23. ↵ Stuart , T. , Butler , A. , Hoffman , P. , Hafemeister , C. , Papalexi , E. , Mauck , W.M. , Hao , Y. , Stoeckius , M. , Smibert , P. , Satija , R. : Comprehensive integration of single-cell data . cell 177 ( 7 ), 1888 – 1902 ( 2019 ) OpenUrl CrossRef PubMed 24. ↵ Student: The probable error of a mean. Biometrika pp. 1 – 25 ( 1908 ) 25. ↵ Svensson , V. : Droplet scrna-seq is not zero-inflated . Nature Biotechnology 38 ( 2 ), 147 – 150 ( 2020 ) OpenUrl CrossRef PubMed 26. ↵ Wolf , F.A. , Angerer , P. , Theis , F.J. : Scanpy: large-scale single-cell gene expression data analysis . Genome biology 19 , 1 – 5 ( 2018 ) OpenUrl CrossRef PubMed View the discussion thread. Back to top Previous Next Posted March 14, 2025. Download PDF Data/Code Email Thank you for your interest in spreading the word about bioRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following LN’s t-Test: A Principled Approach to t-Testing in Single-Cell RNA Sequencing Message Subject (Your Name) has forwarded a page to you from bioRxiv Message Body (Your Name) thought you would like to see this page from the bioRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share LN’s t -Test: A Principled Approach to t -Testing in Single-Cell RNA Sequencing Oskar Kviman , Seong-Hwan Jun , Jens Lagergren bioRxiv 2025.03.12.642799; doi: https://doi.org/10.1101/2025.03.12.642799 Share This Article: Copy Citation Tools LN’s t -Test: A Principled Approach to t -Testing in Single-Cell RNA Sequencing Oskar Kviman , Seong-Hwan Jun , Jens Lagergren bioRxiv 2025.03.12.642799; doi: https://doi.org/10.1101/2025.03.12.642799 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Bioinformatics Subject Areas All Articles Animal Behavior and Cognition (7636) Biochemistry (17697) Bioengineering (13896) Bioinformatics (41957) Biophysics (21457) Cancer Biology (18595) Cell Biology (25522) Clinical Trials (138) Developmental Biology (13381) Ecology (19903) Epidemiology (2067) Evolutionary Biology (24323) Genetics (15612) Genomics (22511) Immunology (17738) Microbiology (40401) Molecular Biology (17184) Neuroscience (88624) Paleontology (667) Pathology (2833) Pharmacology and Toxicology (4825) Physiology (7644) Plant Biology (15158) Scientific Communication and Education (2046) Synthetic Biology (4296) Systems Biology (9825) Zoology (2271)
Text is read by the "Ask this paper" AI Q&A widget below. Extraction quality varies by source — PMC NXML preserves structure cleanly, OA-HTML may include some navigation residue, and OA-PDF can have broken hyphenation. The publisher copy (via DOI) is the canonical version.
My notes (saved in your browser only)
Ask this paper
Answers must be backed by verbatim quotes from this paper's full text. Hallucinated quotes are dropped automatically; if no verbatim passage answers the question, we say so. How this works
Citation neighborhood (sparse)
Too few in-corpus citations on either side for a chart; here are the lists.
Cites (1)
References (25)
- Voyager: exploratory single-cell genomics data analysis with geospatial statistics via crossref
- doi:10.1186/s13059-015-0844-5 via crossref
- doi:10.1038/nbt.4091 via crossref
- doi:10.1016/j.cell.2021.04.048 via crossref
- doi:10.1038/s41587-023-01767-y via crossref
- doi:10.1093/bioinformatics/btn142 via crossref
- doi:10.1038/s41592-019-0619-0 via crossref
- doi:10.1186/s13059-016-1077-y via crossref
- doi:10.1186/s13059-015-0866-z via crossref
- doi:10.1038/nrg2825 via crossref
- doi:10.1186/s13059-014-0550-8 via crossref
- doi:10.1016/j.cell.2015.05.002 via crossref
- doi:10.1111/1467-9469.00115 via crossref
- doi:10.1186/s13059-024-03183-0 via crossref
- doi:10.1093/nar/gkv007 via crossref
- doi:10.1093/bioinformatics/btp616 via crossref
- doi:10.1038/nbt.3192 via crossref
- doi:10.1007/978-0-387-34675-5 via crossref
- doi:10.1016/j.csbj.2020.09.014 via crossref
- doi:10.1038/nmeth.4612 via crossref
- doi:10.1016/j.cell.2019.05.031 via crossref
- doi:10.2307/2331554 via crossref
- doi:10.1038/s41587-019-0379-5 via crossref
- doi:10.1038/nbt.4096 via crossref
- doi:10.1186/s13059-017-1381-1 via crossref
Source provenance
- crossref
- last seen: 2026-06-01T01:00:37.612923+00:00
- europepmc
- last seen: 2026-05-20T01:45:00.602351+00:00
- unpaywall
- last seen: 2026-05-21T05:10:58.409756+00:00
License: CC-BY-4.0