Full text
32,374 characters
· extracted from
preprint-html
· click to expand
Normalization of Single-cell RNA-seq Data Using Partial Least Squares with Adaptive Fuzzy Weight | bioRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-M677548'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search New Results Normalization of Single-cell RNA-seq Data Using Partial Least Squares with Adaptive Fuzzy Weight Vikas Singh , Nikhil Kirtipal , Songwon Lim , Sunjae Lee doi: https://doi.org/10.1101/2024.08.18.608507 Vikas Singh 1 School of Life Sciences , GIST, Gwangju, South Korea Find this author on Google Scholar Find this author on PubMed Search for this author on this site Nikhil Kirtipal 1 School of Life Sciences , GIST, Gwangju, South Korea Find this author on Google Scholar Find this author on PubMed Search for this author on this site Songwon Lim 1 School of Life Sciences , GIST, Gwangju, South Korea Find this author on Google Scholar Find this author on PubMed Search for this author on this site Sunjae Lee 1 School of Life Sciences , GIST, Gwangju, South Korea Find this author on Google Scholar Find this author on PubMed Search for this author on this site For correspondence: leesunjae{at}gist.ac.kr Abstract Full Text Info/History Metrics Preview PDF Abstract Normalization of single-cell RNA-seq (scRNA-seq) is a crucial step in downstream analysis, where raw data are adjusted to correct unwanted factors that prevent the direct comparison of expression measures. scRNA-seq data exhibits a multivariate relationship between transcript-specific expression and sequencing depth that a single scale factor cannot address. A partial least squares (PLS) regression was performed to accommodate the variability of gene expression in each condition, and upper and lower quantiles with adaptive fuzzy weights were utilized to correct unwanted biases in scRNA-seq data. The present approach was compared using real and simulated datasets across various state-of-the-art performance measures. I. I ntroduction Effective pre-processing and normalization play a vital role in studying gene behavior under distinct biological conditions in both scRNA-seq data [ 1 ]–[ 3 ] and high-throughput sequencing data [ 4 ]–[ 6 ]. Gene behavior under these conditions can vary due to unwanted biases caused by transcript length, GC content, dropout, capture efficiency, sequencing depth, and other biological or technical factors [ 7 ]. The implicit normalization approach aids downstream analysis, such as mitigating inflated false positives in differential expression analysis, by correcting of sequencing depth, cell-to-cell differences in capture efficiency, and other technical factors [ 8 ]. In the literature, various methods have been presented for within-sample and between-sample normalization of bulk RNA-seq data [ 9 ]. The most commonly used methods, which have been highly successful, are the trimmed mean of M values (TMM) [ 10 ], DESeq [ 11 ], and median ratio normalization [ 12 ]. However, these methods do not perform well for scRNAseq data due to the prevalence of zero-expression counts. The zero-expression counts make scRNA-seq data extremely sparse, resulting a zero geometric mean (GM) in DESeq and undefined M values in TMM. Additionally, the unbalanced library sizes of abundant and rare cell populations in scRNAseq may violate the non-DE hypothesis, leading to bias [ 13 ]. In recent years, various methods have been proposed for scRNA-seq normalization, falling into two broad classes: global scaling factors and gene-specific scaling factors. The most commonly used approaches for scRNA-seq include multiplying by a constant factor [ 1 ], BASiCS [ 14 ], SAMstrt The code and experimented data are available on the GitHub platform: https://github.com/vikkyak/bIbW/tree/main [ 15 ], the gamma regression model (GRM) [ 16 ], pooling normalization [ 8 ], robust normalization of scRNA-seq data (SCnorm) [ 17 ], and PsiNorm, which models scRNA-seq data using a Pareto distribution [ 18 ]. A significant bias with the global scaling factor is that it accommodates the transcriptspecific expression and sequencing depth relationship by using a common scale factor for all genes in a cell, leading to over-correction when this relationship is rare across genes. However, gene-specific scaling factors can also introduce bias during normalization, particularly when uniform count-depth relationships are assumed across cells, which can vary substantially when marker genes are highly expressed in one cell type but not in others. Recently, kernel-weighted-average robust normalization for single-cell RNA-seq data (scKWARN) [ 13 ] was developed to correct known or unknown technical factors without relying on explicit count-depth relationships or data distributions. It generates a pseudo-expression count for each cell using fuzzy technical neighbors through kernel smoothing. The present approach overcomes biases due to library size, dropout, RNA composition, and other technical factors and is motivated by two different methods: pooling normalization [ 8 ], which pools cells with similar library sizes, and scKWARN [ 13 ], which does not rely on specific count-depth relationships or data distributions. In this approach, the scRNAseq data is clustered, and PLS analysis is performed on each cluster or condition. To preserve data variability, correlation is calculated for each condition, and each condition is multiplied by the average correlation to minimize differences in library size before normalization. Each condition is then divided into upper and lower quantiles using the mid-quantile function of Qtools, and a confidence interval is obtained using the confint.midquantile function. PLS analysis is conducted using the pls package in R to obtain the first component from the upper and lower quantiles that exhibits highest variance. This first component is used to measure the adaptive fuzzy weight of the upper and lower quantiles to determine the expression level of the genes, which is then used to calculate the scale factor and normalize the data, as described in Figure 1 . Download figure Open in new tab Fig. 1: Overview of sample normalization of scRNA-seq data using partial least square with adaptive Gaussian fuzzy weight. The rest of the paper is organized as follows: Methods of the present approach is discussed in Section II. Results on real and simulated data is discussed in Section III. Computational performance in IV, Finally, Section V concludes the paper. II. M aterial AND M ethods The present approach first utilizes unsupervised clustering to pool the cells. Then, without relying on specific countdepth relationships or data distributions, it generates a pseudoexpression count for each cell. This approach is divided into five distinct steps as follows: In the first step, if the conditions of datasets are not available. The present approach, cluster the datasets using quickCluster function of scran library in R to group the dataset into multiple clusters. However, any clustering algorithm can be utilized to group the datasets into a random number of groups to proceed to the next step. The number of variables (genes) in each group may differ, and the nature of the variables (quantitative or qualitative) can vary from one group to another, but the variables should be of the same type within a given group [ 19 ]. To address this issue, we used Partial Least Squares (PLS) regression [ 20 ], which simultaneously considers multiple sets of variables and balances the influence of each set. We ran pls with one component and calculated the cross-correlations between the components associated with each condition. To balance each group or condition, we multiplied it by the average cross-correlation. In the third step, technical similarity for each condition is estimated using the first PLS component with 25 th , 50 th , and 75 th percentiles of the upper and lower quantiles of non-zero expression count in cells, using the midquantile function from the Qtools package in R. To elaborate, assume an scRNAseq data has m genes and n cells. Let y gj be the expression count of g gene in the j cell for g = 1, …, m , and j = 1, …, n . Define P j = {g : y gj > 0, g = 1, …, m} is the subset of genes with non-zero expression counts in cell j . Let Ql j = [ q 0.25, j , q 0.50, j , q 0.75, j ] and Qu j = [ q 0.25, j , q 0.50, j , q 0.75, j ] represent the upper and lower log nonzero expression counts, respectively, with each being a 3-D vector containing 25 th , 50 th , and 75 th percentiles of non-zero expression counts with log-transformed in cell j . Let dl = [ d 1 , d 2 , …, d n ] and du = [ d 1 , d 2 , …, d n ] are the first PLS component of Ql and Qu , respectively. In the fourth step dl and du are utilized to measure the technical variability among cells by estimating the weight and as shown in Figure 1 , step, Gaussian membership function. The lower and upper weights are defined as follows: Where K(.) is the Gaussian kernel function, and b g is bandwidth eastimated by a kernel density estimate in R package KernSmooth. The pseudo expression count for cell j are written as Where 5: The final step creates a reference expression count for each cell j , denoted as M j : Where Here, m g is the geometric mean (GM) of gene g across the subset of cells R g where the gene is expressed. In other words, for each non-zero count gene g in cell j , the process involves determining the subset of cells express this gene ( R g ) and finding the GM of the gene expression across R g . It is important to note that M j is calculated based on non-zero count of genes in the cell j , making M j specific to cell j . The pseudo count of each cell is then compared to its specific reference count to find the size factor of each cell. The present approach estimates the normalization factor using a median ratio, as follows: III. R esults AND D iscussion The performance of the proposed (Pro) approach is assessed in terms of F1 score, log-fold change, root mean square error (RMSE), and the proportion of differentially expressed genes, and is compared with other state-of-the-art methods, including Relative Counts (RC), scran, sctransform, scKWARN, and PsiNorm, using three real scRNA-seq datasets and two simulated datasets described below. A. Real Datasets 1) PBMC33K The first dataset used in this study is PBMC33K, generated by 10x Genomics, which comprises 33,148 human peripheral blood mononuclear cells and 32,738 genes. Similar to scKWARN, we randomly extracted 1,000 B cells and generated two cell populations for a comparable analysis. A fixed percentage of genes was randomly selected and scaled with distinct factors to increase the proportion of DE genes in these two cell populations. The present approach was tested on moderate and strong DE genes with balanced and unbalanced cell populations. In the case of moderate DE, the percentage of genes varied from 5% to 15% in each cell population; in strong DE, the percentage of genes varied from 15% to 30%. The balanced case had 1,000 cells in each population, while the unbalanced case had one population with 1,000 cells and the other with 400 cells. Model-based analysis of single-cell transcriptomics ( MAST ) was utilized to identify the DE genes [ 21 ]. In all cases, the present approach was compared with scKWARN, RC, sctransform, scran, and PsiNorm using performance metrics such as the F1 score and RMSE with true fold change. For moderate DE with balanced and unbalanced settings, as shown in Figure 2 , the present approach achieved comparable or better performance in F1 scores and RMSE values. Figure 2 also shows that the present approach accurately estimates the true log-fold change while varying the percentage of DE genes within the cell population. For strong DE with balanced and unbalanced settings, as shown in Figure 3 , the present approach achieved performance comparable to scKWARN and outperformed the other normalization methods in F1 scores and RMSE values. Download figure Open in new tab Fig. 2: Performance analysis with state-of-the-art on PBMC33K dataset with Moderate (10%) Differentially Expressed Genes. Download figure Open in new tab Fig. 3: Performance analysis with state-of-the-art on PBMC33K dataset with Strong (30%) Differentially Expressed Genes. 2) GSE29087 The second dataset used in this study was accessed from the Gene Expression Omnibus (GEO) database with accession number GSE29087. It contains 92 cells, comprising 48 mouse embryonic cells and 44 mouse embryonic fibroblast cells, with 22,928 genes. Genes with low expression counts across the libraries provide little information for differential expression (DE) analysis. These genes were filtered out if they were expressed in fewer than three cells. After removing the low-expression genes, 7,061 genes were included for DE analysis. Among these genes, 718, validated through qRT-PCR analysis, were taken as the gold standard DE genes, representing only a subset of true DE genes. In the current study, we used the top G genes ranked by each method as the gold standard genes. The performance was compared in three scenarios: none of cells, half of the cells, and all of cells downsized using a uniform random distribution0020with a minimum of 0.2 and a maximum of 1. When no cells were downsized and when half of the cells were randomly downsized, sctransform performed best in detecting DE genes, followed by the proposed approach, as shown in Figure 4 . scran had the lowest proportion, possibly due to the limited number of cells, which may not be sufficient for accurate estimation of the scale factor. When all cells were downsized, the proposed approach again achieved better detection of DE genes compared to the other methods, except for sctransform, as shown on the right side of Figure 4 . Download figure Open in new tab Fig. 4: Performance analysis of proposed approach (Pro) with state-of-the-art on the GSE29087 dataset. Y axis represents the proportion of 718 gold standard genes in the top G genes on X axis. The performance was compared in three scenario: left plot with none of cells downsized,, middle plot with half of the cells downsized, and right plot with all cells downsized. 3) GSE60361 The third dataset was accessed from GEO database with accession number GSE60361. It contains data from the somatosensory cortex (S1) and hippocampus CA1 area of juvenile (P22-P32) CD1 mice, comprising 33 males and 34 females. Cells were collected without selection, except for 116 cells obtained by FACS from 5HT3a-BACEGFP transgenic mice. A total of 76 Fluidigm C1 runs were performed, each attempting 96 cell captures and resulting in 3,005 highquality single-cell cDNAs, sequenced at an average depth of 14,000 reads per cell (with UMI) [ 22 ]. Similar to the first study, the present approach was tested on moderate DE and strong DE with balanced and unbalanced cell populations. For moderate DE, the percentage of genes varied from 5% to 15%; however, in strong DE, the percentage of genes varied from 15% to 30%. The balanced case involved each cell population having 1,000 cells, while the unbalanced case had one population with 1,000 cells and the other with 400 cells, similar to Study 1. In all cases, the present approach was compared with scKWARN, scran, RC, sctransform, and PsiNorm using performance metrics such as the F1 score and RMSE with true fold change. For moderate DE with balanced and unbalanced cases, as shown in Figure 5 , the present approach achieved comparable or better performance in F1 scores and RMSE values. Figure 5 also shows that the present approach accurately estimates the true log-fold change while varying the percentage of DE genes within the cell population. For strong DE with balanced and unbalanced cases, as shown in Figure 6 , the performance of the present approach was comparable to scKWARN and outperformed the rest of the normalization methods in F1 scores and RMSE values. Download figure Open in new tab Fig. 5: Performance analysis with state-of-the-art on GSE60361 dataset with Moderate (10%) Differentially Expressed Genes. Download figure Open in new tab Fig. 6: Performance analysis with state-of-the-art on GSE60361 dataset with Strong (30%) Differentially Expressed Genes. B. Simulated Datasets The simulated data were generated using a negative binomial distribution under two cases: a linear relationship between the mean and count-depth [ 1 ], and no linear relationship between the mean and count-depth [ 8 ]. One hundred simulated datasets were generated for each case, consisting of 3,000 genes across three distinct cell groups. The effects of noise, such as library size, dropout rate, and RNA composition, were studied in both balanced and unbalanced settings. The simulated data were also analyzed by varying the percentage of DE genes, ranging from moderate to strong DE. 1) Simulation I (SIM-I) The first simulation setting, y gj , is generated from a negative binomial distribution with dispersion 0.1 and mean: Where The parameter, ϕ g = 1 is chosen for all genes excluding the DE genes in each of the three distinct cell groups. For DE genes, ϕ g = 8 p is chosen, where p follow the Bernoulli distribution: For altering library sizes of cells, we considered: In the case of RNA compositions, 5% genes are randomly chosen (excluding DE genes) and assumed to be highly expressed genes by multiplying their expression levels with a positive constant chosen from the set { 12, 13, …, 20 } . These 5% highly expressed genes dominate 20% to 60% of library sizes. To model dropout rates, we increased the dropout with an additional probability of 10% to 30% as follows: Where 2) Simulation II (SIM-II) In the second simulation setting, y gj , is also generated from negative binomial distribution with mean µ gj as of (8) with similar dispersion value of 0.1. The parameter λ g follow a gamma distribution, specifically: The parameter, ϕ g is similar to that in condition SIM I. In the case of altering library sizes of cells, the mean of the negative binomial distributions were set as: Where In the case of RNA compositions, 5% genes are randomly chosen (excluding DE genes) similar to SIM-I and assumed as highly expressed genes by multiplying a positive constant chosen from the set { 12, 13, …, 20 } with probability of 1 / 9. The 5% highly expressed genes dominate 20% to 60% of library sizes. For dropout rates, we also increased dropout with an additional probability 10% to 30% as of (13) in SIM-I. The performance on the simulated datasets was assessed using four different metrics: F1 score, sensitivity, specificity, and bias. The R package MAST was used to identify DE genes, and the difference between the true and estimated log 2 fold change was measured in terms of bias. The F1 score provides a balance between precision and recall. Additionally, sensitivity and specificity were used to assess the performance of the proposed approach. The performance was evaluated in terms of varying library size, RNA composition, and dropout for SIM-I and SIM-II. Figure 7 shows the performance of the present approach compared with state-of-the-art methods by altering the library size. As shown in Figure 7 , the present approach exhibited less bias in both SIM-I and SIM-II, with improved overall performance. In the case of RNA composition, as shown in Figure 8 , the present approach achieved a better F1 score, indicating a superior balance between precision and recall in both SIM-I and SIM-II. Finally, performance was evaluated based on dropout rate, as shown in Figure 9 , for both SIM-I and SIM-II on moderately DE genes. Download figure Open in new tab Fig. 7: Performance analysis with state-of-the-art on the library size of dataset with Moderate Differentially Expressed Genes: (a) F1 score (left SIM-I and right SIM-II), (b) Sensitivity plot (left SIM-I and right SIM-II), (c) Specificity plot (left SIM-I and right SIM-II), and (d) Bias plot (left SIM-I and right SIM-II). Navie (RC), scK (scKWARN), sct(sctransform) and Psi(PsiNorm) Download figure Open in new tab Fig. 8: Performance analysis with state-of-the-art on the RAN composition with Moderate Differentially Expressed Genes: (a) F1 score (left SIM-I and right SIM-II), (b) Sensitivity plot (left SIM-I and right SIM-II), (c) Specificity plot (left SIM-I and right SIM-II), and (d) Bias plot (left SIM-I and right SIM-II). Download figure Open in new tab Fig. 9: Performance comparison on the Dropout with Moderate Differentially Expressed Genes: (a) F1 score (left SIM I and right SIM II), (b) Sensitivity plot (left SIM-I and right SIM-II), (c) Specificity plot (left SIM-I and right SIM-II), and (d) Bias plot (left SIM-I and right SIM-II). The comprehensive study of both real and simulated data shows that scran and PsiNorm have high RMSE when cell populations were unbalanced in both moderate and strong DE levels. The proposed approach and scKWARN consistently achieved low RMSE values across all three settings when biases were introduced by altering the library size, RNA composition, and dropout rates. sctransform estimates genespecific scaling factors, and its performance highly depends on the fitting of the model and whether the simulation settings align with their underlying assumptions. IV. C omputational P erformance The computational performance of the present approach is presented in Table I . The execution time of the present approach is comparable to state-of-the-art methods. sctransform has a longer computation time because it estimates the genespecific scale factor by explicitly fitting each gene with a generalized linear model. View this table: View inline View popup Download powerpoint TABLE 1: C omparison of R untime (I n S econds ) V. C onclusion This paper presents a robust normalization method for scRNA-seq data using PLS regression. Variability between conditions was minimized by multiplying with the average correlation. Biases in library size were reduced by modeling the scale factor with adaptive fuzzy weights using the upper and lower quintiles of the data. The effectiveness of the proposed approach was validated using real and simulated datasets and was compared with state-of-the-art methods across various performance metrics. The results show that the proposed approach effectively corrects biases due to library size, RNA composition, and dropout in scRNA-seq datasets. Footnotes vikkyak07{at}gist.ac.kr kirtipal.n{at}gmail.com songwon409{at}gm.gist.ac.kr leesunjae{at}gist.ac.kr R eferences [1]. ↵ Christoph Hafemeister and Rahul Satija . Normalization and variance stabilization of single-cell RNA-seq data using regularized negative binomial regression . Genome biology , 20 ( 1 ): 296 , 2019 . OpenUrl CrossRef PubMed [2]. Rory Stark , Marta Grzelak , and James Hadfield . RNA sequencing: the teenage years . Nature Reviews Genetics , 20 ( 11 ): 631 – 656 , 2019 . OpenUrl [3]. ↵ Adrienne Niederriter et al. , Shami. Single-cell RNA sequencing of human, macaque, and mouse testes uncovers conserved and divergent features of mammalian spermatogenesis . Developmental cell , 54 ( 4 ): 529 – 547 , 2020 . OpenUrl CrossRef PubMed [4]. ↵ Vikas Singh , Nishchal K Verma , and Yan Cui . Type-2 fuzzy PCA approach in extracting salient features for molecular cancer diagnostics and prognostics . IEEE Transactions on Nanobioscience , 18 ( 3 ): 482 – 489 , 2019 . OpenUrl [5]. Vikas Singh and Nishchal K Verma . Gene expression data analysis using feature weighted robust fuzzy-means clustering . IEEE Transactions on NanoBioscience , 22 ( 1 ): 99 – 105 , 2022 . OpenUrl [6]. ↵ Sevakula Rahul K et al. Transfer learning for molecular cancer classification using deep neural networks . IEEE/ACM transactions on computational biology and bioinformatics , 16 ( 6 ): 2089 – 2100 , 2018 . OpenUrl [7]. ↵ Ciaran Evans , Johanna Hardin , and Daniel M Stoebel . Selecting between-sample RNA-seq normalization methods from the perspective of their assumptions . Briefings in bioinformatics , 19 ( 5 ): 776 – 792 , 2018 . OpenUrl [8]. ↵ Aaron T L. Lun , Karsten Bach , and John C Marioni . Pooling across cells to normalize single-cell RNA sequencing data with many zero counts . Genome biology , 17 : 1 – 14 , 2016 . OpenUrl CrossRef PubMed [9]. ↵ Vikas Singh , Nikhil Kirtipal , Byeongsop Song , and Sunjae Lee . Normalization of RNA-seq data using adaptive trimmed mean with multi-reference . Briefings in Bioinformatics , 25 ( 3 ): bbae241 , 2024 . OpenUrl [10]. ↵ Mark D Robinson and Alicia Oshlack . A scaling normalization method for differential expression analysis of RNA-seq data . Genome biology , 11 : 1 – 9 , 2010 . OpenUrl CrossRef [11]. ↵ Michael I Love , Wolfgang Huber , and Simon Anders . Moderated estimation of fold change and dispersion for RNA-seq data with DESeq2 . Genome biology , 15 : 1 – 21 , 2014 . OpenUrl CrossRef PubMed [12]. ↵ Elie Maza , Pierre Frasse , Pavel Senin , Mondher Bouzayen , and Mohamed Zouine . Comparison of normalization methods for differential gene expression analysis in RNA-seq experiments: a matter of relative size of studied transcriptomes . Communicative & integrative biology , 6 ( 6 ): e25849 , 2013 . OpenUrl CrossRef [13]. ↵ Borella Matteo et al. PsiNorm: a scalable normalization for single-cell RNA-seq data . Bioinformatics , 38 ( 1 ): 164 – 172 , 2022 . OpenUrl [14]. ↵ Catalina A Vallejos , John C Marioni , and Sylvia Richardson . BASiCS: Bayesian analysis of single-cell sequencing data . PLoS computational biology , 11 ( 6 ): e1004333 , 2015 . OpenUrl [15]. ↵ Shintaro Katayama , Virpi Töhönen , Sten Linnarsson , and Juha Kere . SAMstrt: statistical test for differential expression in single-cell tran-scriptome with spike-in normalization . Bioinformatics , 29 ( 22 ): 2943 – 2945 , 2013 . OpenUrl CrossRef PubMed [16]. ↵ Bo Ding et al. Normalization and noise reduction for single cell RNA-seq experiments . Bioinformatics , 31 ( 13 ): 2225 – 2227 , 2015 . OpenUrl CrossRef PubMed [17]. ↵ Rhonda et al. , Bacher. SCnorm: robust normalization of single-cell RNA-seq data . Nature methods , 14 ( 6 ): 584 – 586 , 2017 . OpenUrl [18]. ↵ Chih-Yuan Hsu , Chia-Jung Chang , Qi Liu , and Yu Shyr . scKWARN: Kernel-weighted-average robust normalization for single-cell RNA-seq data . Bioinformatics , 40 ( 2 ): btae008 , 2024 . OpenUrl [19]. ↵ Hervé Abdi and Lynne J Williams . Principal component analysis . Wiley interdisciplinary reviews: computational statistics , 2 ( 4 ): 433 – 459 , 2010 . OpenUrl CrossRef [20]. ↵ Hervé Abdi , Dominique Valentin , et al. Multiple factor analysis (mfa) . Encyclopedia of measurement and statistics , pages 657 – 663 , 2007 . [21]. ↵ Finak Greg et al. MAST: a flexible statistical framework for assessing transcriptional changes and characterizing heterogeneity in single-cell RNA sequencing data . Genome biology , 16 : 1 – 13 , 2015 . OpenUrl CrossRef PubMed [22]. ↵ Zeisel et al. Cell types in the mouse cortex and hippocampus revealed by single-cell RNA-seq . Science , 347 ( 6226 ): 1138 – 1142 , 2015 . OpenUrl Abstract / FREE Full Text View the discussion thread. Back to top Previous Next Posted August 19, 2024. Download PDF Email Thank you for your interest in spreading the word about bioRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following Normalization of Single-cell RNA-seq Data Using Partial Least Squares with Adaptive Fuzzy Weight Message Subject (Your Name) has forwarded a page to you from bioRxiv Message Body (Your Name) thought you would like to see this page from the bioRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share Normalization of Single-cell RNA-seq Data Using Partial Least Squares with Adaptive Fuzzy Weight Vikas Singh , Nikhil Kirtipal , Songwon Lim , Sunjae Lee bioRxiv 2024.08.18.608507; doi: https://doi.org/10.1101/2024.08.18.608507 Share This Article: Copy Citation Tools Normalization of Single-cell RNA-seq Data Using Partial Least Squares with Adaptive Fuzzy Weight Vikas Singh , Nikhil Kirtipal , Songwon Lim , Sunjae Lee bioRxiv 2024.08.18.608507; doi: https://doi.org/10.1101/2024.08.18.608507 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Bioinformatics Subject Areas All Articles Animal Behavior and Cognition (7651) Biochemistry (17746) Bioengineering (13928) Bioinformatics (42064) Biophysics (21499) Cancer Biology (18650) Cell Biology (25579) Clinical Trials (138) Developmental Biology (13409) Ecology (19947) Epidemiology (2067) Evolutionary Biology (24373) Genetics (15633) Genomics (22557) Immunology (17774) Microbiology (40504) Molecular Biology (17217) Neuroscience (88793) Paleontology (667) Pathology (2845) Pharmacology and Toxicology (4836) Physiology (7664) Plant Biology (15178) Scientific Communication and Education (2047) Synthetic Biology (4304) Systems Biology (9839) Zoology (2272)
Text is read by the "Ask this paper" AI Q&A widget below.
Extraction quality varies by source — PMC NXML preserves structure
cleanly, OA-HTML may include some navigation residue, and OA-PDF can
have broken hyphenation. The publisher copy
(via DOI)
is the canonical version.