Cell type–specific functions of nucleic acid-binding proteins revealed by deep learning on co-expression networks

preprint OA: closed
📄 Open PDF Full text JSON View at publisher

Abstract

Understanding how regulatory architectures are reorganized during biological state transitions remains a central challenge in functional genomics. Here, we integrate co-expression–derived regulatory interactions with interpretable deep learning to compute gene-level contribution scores and introduce ΔNES (normalized enrichment score difference) to quantify pathway redistribution across biological states. Applying this framework to neural progenitor and leukemia-associated cellular states, we identify systematic redistribution of functional modules across RNA-binding proteins, including PKM, HNRNPK, and NELFE. Neural System– and Immune System–associated modules are differentially positioned along contribution-ranked regulatory landscapes, while Signal Transduction consistently forms a conserved signaling backbone. These findings suggest that pathway redistribution reflects regulatory reorganization across biological states and provides a framework for interpreting large-scale regulatory coordination beyond expression-centric analyses.
Full text 102,724 characters · extracted from preprint-html · click to expand
Pathway redistribution reveals a shared signaling backbone and context-dependent regulatory modules in RNA-binding protein networks | bioRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-M677548'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search New Results Pathway redistribution reveals a shared signaling backbone and context-dependent regulatory modules in RNA-binding protein networks View ORCID Profile Naoki Osato , View ORCID Profile Kengo Sato doi: https://doi.org/10.1101/2025.03.03.641203 Naoki Osato 1 School of Life Science and Technology, Institute of Science Tokyo , Tokyo, Japan Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Naoki Osato For correspondence: naokiosato11{at}gmail.com Kengo Sato 1 School of Life Science and Technology, Institute of Science Tokyo , Tokyo, Japan Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Kengo Sato Abstract Full Text Info/History Metrics Supplementary material Data/Code Preview PDF Abstract Understanding how regulatory architectures are reorganized across cellular contexts remains a central challenge in functional genomics. Here, we integrate co-expression–derived candidate regulatory interactions with interpretable deep learning to generate gene-level contribution scores and introduce ΔNES (normalized enrichment score difference) to quantify pathway redistribution between cellular states. Because gene expression reflects the combined effects of multiple regulatory inputs, contribution scores capture relative regulatory influence rather than transcriptional abundance itself. Applying this framework to neural progenitor cells and K562 leukemia cells, we identify systematic redistribution of functional modules across multiple RNA-binding proteins, including PKM, HNRNPK, and NELFE. Neural System– and Immune System–associated modules are differentially positioned along the ΔNES spectrum, indicating context-dependent redistribution of regulatory influence rather than isolated pathway activation events. At the pathway level, Signal Transduction consistently forms a shared signaling backbone across proteins and cellular contexts, while modules related to neuronal functions, immune responses, and developmental processes exhibit context-dependent redistribution. Subpathway analysis further reveals convergence on receptor-mediated signaling processes, including FGFR/RTK-, IRS-, and MAPK-related pathways. These redistribution patterns are preserved under alternative DeepLIFT background settings despite polarity changes in contribution–expression correlations, indicating that pathway-level contrasts arise from stable rank-structure differences rather than background-dependent score artifacts. Together, our findings demonstrate that contribution score–based pathway ranking reveals a conserved signaling backbone alongside context-dependent functional modules, providing a framework for interpreting regulatory architecture beyond expression-centric analyses. Introduction Understanding how regulatory architectures are organized and reconfigured across cellular contexts remains a central challenge in molecular biology 1 – 10 . Nucleic acid-binding proteins (NABPs), including DNA-binding proteins and RNA-binding proteins (RBPs), play key roles in gene regulation; however, their functional characterization across diverse cell types remains incomplete. While experimental approaches such as ChIP-seq and eCLIP can identify binding sites, these methods are not easily scalable and often fail to capture indirect or context-dependent regulatory interactions 11 , particularly for RBPs lacking well-defined binding motifs 12 – 14 . Computational approaches, including deep learning models, have improved the prediction of gene expression by integrating regulatory features derived from binding data. However, these methods typically rely on experimentally defined binding sites or sequence features and do not directly reveal how individual NABPs contribute to gene expression programs at the systems level. In particular, it remains unclear how regulatory influence is organized across pathways and how it is reconfigured between cellular contexts. Gene co-expression analysis provides a complementary perspective by capturing coordinated expression patterns across diverse biological conditions, reflecting both direct and indirect regulatory relationships 15 – 23 . Integrating such information into interpretable deep learning frameworks offers the potential to quantify regulatory influence at the gene level without requiring prior knowledge of binding motifs or direct interaction data. Here, we integrate co-expression–derived regulatory interactions with interpretable deep learning to compute gene-level contribution scores and introduce ΔNES (normalized enrichment score difference) to quantify pathway redistribution between cellular contexts. Rather than focusing on individual regulatory interactions, this framework enables the analysis of regulatory architecture at the pathway level. Applying this approach to neural progenitor cells and K562 leukemia cells, we find that regulatory influence is not organized as distinct pathway sets, but instead is structured around a shared signaling backbone, while functional modules are redistributed in a context-dependent manner. This conceptual framework provides a systems-level perspective on gene regulation, revealing how common signaling networks are differentially utilized across cellular states. Results Gene co-expression features enhance expression prediction accuracy Replacing low-contribution DBP inputs with gene co-expression data led to a substantial increase in model performance ( Fig. 1 ). In human foreskin fibroblasts (HFFs), substituting 302 out of 1,310 DBPs with co-expression profiles improved the correlation between predicted and actual gene expression from 0.70 to 0.80 ( Fig. 2a, b ). These co-expression inputs included both NABP-encoding and non-NABP mRNAs, with gene symbols for DNA- and RNA-binding proteins obtained from the ENPD (Eukaryotic Nucleic acid Binding Protein) database 24 , and co-expression profiles sourced from COXPRESdb 15 . A subsequent experiment replaced 325 low-contribution DBPs with co-expression interactions involving RBPs, resulting in a further increase in prediction accuracy ( r = 0.81; Fig. 2c ). Collectively, these findings demonstrate that integrating co-expression interactions into the model enhances its ability to capture gene regulatory dynamics beyond direct binding site information. Download figure Open in new tab Figure 1. Overview of the deep learning framework for predicting and functionally characterizing NABP regulatory targets. a. This diagram outlines the overall strategy used to infer the functional roles of nucleic acid-binding proteins (NABPs) from gene expression data. Gene expression profiles from various human cell types were integrated with NABP–gene co-expression relationships derived from COXPRESdb. These inferred interactions were used to replace low-contribution DNA-binding proteins in a deep learning model originally trained on ChIP-seq–based binding site data. The modified model was trained to predict gene expression levels and identify regulatory target genes using DeepLIFT-derived importance scores. Functional annotations of the predicted target genes were assessed using conventional enrichment analyses and GSEA, along with a ChatGPT-based method for literature-guided interpretation. b. Gene expression levels in human cells were predicted using the genomic locations of DNA-binding sites at promoters and distal elements. During model training, a DNA-binding protein bound to a distal region overlapping an expression-associated SNP (eSNP) was linked to a potentially regulated gene (eGene) based on eQTL data. For each gene–protein pair, ChIP-seq peak locations were binned (indicated by green boxes) and formatted into a matrix as input for deep learning. In co-expression–based models, binding sites were artificially assigned near transcription start sites (TSSs). The deep learning model consisted of 1D convolutional layers, global average pooling, and fully connected layers. It predicted gene expression levels across 25,071 genes and multiple human cell types. Download figure Open in new tab Download figure Open in new tab Figure 2. Relationship between gene expression levels and contribution scores. Gene expression levels (log₂ TPM) predicted by a deep learning model were compared with observed expression levels using Spearman’s rank correlation. a–c . Schematic overview of input feature configurations used for gene expression prediction. a. DNA-binding sites of 1,310 proteins derived from ChIP-seq data. b. Replacement of 302 low-contribution DNA-binding proteins with co-expression–derived DNA-binding proteins. c. Replacement of 325 DNA-binding proteins with RNA-binding proteins identified from co-expression relationships. d. Comparison of absolute DeepLIFT contribution scores for predicted regulatory sites derived from gene co-expression versus ChIP-seq data, using a background defined as 0.5× the original input feature values. Each pair of violin plots represents co-expression–based sites (right) and corresponding ChIP-seq sites not included in that set (left). Contribution scores were significantly higher for co-expression–based sites (Wilcoxon signed-rank test, p -values < 10⁻⁶). Results obtained using an amplified background (input × 2) are shown in Supplementary Fig. S2c . e. f. Relationship between gene expression levels and DeepLIFT contribution scores for representative DNA- and RNA-binding proteins across four cell types. Contribution scores were computed using a background defined as input × 0.5 ( e ) or input × 2 ( f ). The same set of proteins is shown in both panels, enabling direct comparison of background effects. While background scaling alters the sign of the DeepLIFT–expression correlation, the relative ordering of contribution scores and cell type–specific patterns are preserved. For each protein, the Spearman correlation coefficient and corresponding p -value displayed above the plot correspond to the cell type showing the strongest correlation among the four analyzed cell types. Data points are color-coded by cell type, revealing consistent distributional trends across all four cell types. Contribution scores derived from co-expression–based NABP–gene interactions exhibited broader distributions and higher overall magnitudes than those obtained from ChIP-seq–based inputs across all four cell types examined (Wilcoxon signed-rank test, p < 10⁻ 2 ; Fig. 2d and Supplementary Fig. S2c ; Supplementary Table S1 ). Each pair of violin plots represents absolute contribution scores for co-expression–based sites (right plot in each pair) and the corresponding ChIP-seq sites not included in that set (left plot). A – P correspond to specific cell types and NABP classes as follows: B , DBP co-expression–based sites in neural progenitor cells (NPCs); A , NPC ChIP-seq sites excluding B . D , RBP co-expression–based sites in NPCs; C , NPC ChIP-seq sites excluding D . F , DBP co-expression–based sites in HepG2 cells; E , HepG2 ChIP-seq sites excluding F . H , RBP co-expression–based sites in K562 cells; G , K562 ChIP-seq sites excluding H . J , DBP co-expression–based sites in HMEC cells; I, HMEC ChIP-seq sites excluding J . L , RBP co-expression–based sites in HMEC cells; K , HMEC ChIP-seq sites excluding L . N , DBP co-expression–based sites in human foreskin fibroblasts (HFFs); M , HFF ChIP-seq sites excluding N . P , RBP co-expression–based sites in HFFs; O , HFF ChIP-seq sites excluding P . Corresponding violin plots generated using an amplified background (input × 2) are shown in Supplementary Figure S2 c . However, the nature of this difference depended on the class of nucleic acid–binding protein. For DNA-binding proteins (DBPs), contribution score distributions differed significantly between ChIP-seq–derived and co-expression–derived regulatory sites, yet their mean and median values were often comparable across cell types. This indicates that, for DBPs, co-expression–based feature selection primarily reshapes the distributional structure of contribution scores without uniformly shifting their central tendency. In contrast, for RNA-binding proteins (RBPs), regulatory sites inferred from gene co-expression consistently showed substantially higher DeepLIFT contribution scores than those derived from experimental binding data alone. This pattern was observed across all examined cell types, including K562, NPCs, HMECs, and HFFs, and was reflected by marked increases in both mean and median contribution scores. These results suggest that co-expression–based inference preferentially enriches for RBP-associated regulatory interactions that contribute more directly to gene expression prediction, whereas experimental binding data for RBPs may include a larger fraction of interactions that are weakly coupled to transcriptomic output. These observations are consistent with previous reports showing that incorporating RNA-binding protein– and miRNA-associated features improves gene expression prediction accuracy, as exemplified by the DEcode framework 25 . Importantly, these trends were robust to DeepLIFT background definition: comparable differences between ChIP-seq– and co-expression–derived targets were observed under both background scaling schemes (input × 0.5 and input × 2). Thus, while background scaling influences the sign and absolute scale of contribution scores, it does not alter the fundamental distinction between binding-based and co-expression–based regulatory inference. To assess the influence of background selection on DeepLIFT-derived contribution scores, we compared results obtained using a reduced background (input × 0.5; Fig. 2e ) and an amplified background (input × 2; Fig. 2f ) for the same representative DNA- and RNA-binding proteins. Under the reduced background, contribution scores tended to show negative correlations with gene expression, whereas the amplified background produced predominantly positive correlations. These patterns appeared approximately mirrored around zero, consistent with DeepLIFT’s definition as a relative contribution measure with respect to the chosen reference. Importantly, despite this inversion in sign, the relative ranking of genes and the overall cell type–specific structure of contribution score distributions remained highly consistent across background settings. In both cases, genes with larger-magnitude contribution scores were preferentially associated with higher or lower expression levels depending on the background, indicating that DeepLIFT captures relative regulatory importance rather than absolute activation or repression. A conceptual model explaining how input definition and background scaling give rise to these patterns is provided in Supplementary Fig. S4 . Notably, background-dependent differences were more pronounced for RNA-binding proteins (RBPs) ( Fig. 2e,f ). For the same representative RBPs, scatter plots generated using the reduced background (input × 0.5) exhibited more distinct, partially separated distributions across the four cell types, whereas under the amplified background (input × 2), data points from different cell types overlapped more extensively. This suggests that the reduced-background configuration enhances sensitivity to cell type–specific regulatory variation for RBPs. In contrast, DNA-binding proteins (DBPs) generally showed stronger correlations with gene expression under both background conditions, consistent with their more direct transcriptional regulatory roles. Together, these results underscore the importance of background-aware and protein class–specific interpretation of DeepLIFT contribution scores. Experimental binding data support model-inferred regulatory targets To determine whether DeepLIFT-derived regulatory contribution scores reflect biologically meaningful nucleic acid–binding protein (NABP) activity, we systematically compared model-inferred regulatory targets with experimentally validated ChIP-seq and eCLIP binding data across multiple genomic regions, threshold definitions, and cellular contexts ( Fig. 3 ). Experimentally validated binding sites are consistent with deep learning–derived regulatory targets. Genes with ChIP-seq or eCLIP peaks for a given NABP exhibited disproportionately high or low importance ranks, indicating that experimentally confirmed binding sites are associated with stronger regulatory influence in gene expression prediction models. This relationship supports the biological validity of DeepLIFT-based contribution scores, which quantify the importance of specific NABPs in driving gene expression changes. Download figure Open in new tab Download figure Open in new tab Download figure Open in new tab Figure 3. Comparison of predicted regulatory target genes with experimental ChIP-seq and eCLIP binding data. This figure illustrates the correspondence between gene-level regulatory contribution scores inferred by deep learning and experimentally identified nucleic acid-binding protein (NABP) binding sites, using a DeepLIFT background defined as 0.5× the original input feature values. a. Rank distribution of DeepLIFT contribution scores for nucleic acid-binding sites validated by ChIP-seq (DNA-binding proteins) or eCLIP (RNA-binding proteins). Contribution scores were ranked from highest to lowest and grouped into equal-sized rank bins. Binding sites located within promoters or transcribed regions of experimentally bound genes showed non-random enrichment toward higher or lower contribution ranks compared with all binding sites. b. Proportion of high-contribution genomic regions overlapping ChIP-seq binding sites in promoter-associated regions. The fractions of high-contribution regions overlapping ChIP-seq peaks are shown for promoter + enhancer regions (±100 kb from the transcription start site) and promoter-only regions (±5 kb). Multiple rank- and score-based thresholds were applied to define high- and low-contribution regions (see Supplementary Note and Tables S2 and S3 ). c. Overlap between high-contribution regions and eCLIP peaks for RNA-binding proteins. Cumulative distributions show the proportion of high-contribution gene regions overlapping experimentally determined eCLIP peaks within transcribed regions (from transcription start site to transcription end site), compared with random controls. Corresponding analyses using an amplified DeepLIFT background (input × 2) produced similar results and are shown in Supplementary Figure S3 . d. Comparison of DeepLIFT contribution scores between HepG2 and K562 cells under different background definitions. Scatter plots display contribution scores for predicted regulatory target genes of three representative NABPs (RPLP0, ETF1, and NELFE), comparing K562 (x-axis) and HepG2 (y-axis). Absolute contribution scores were log₂-transformed for visualization, and only genes with experimentally validated binding sites were included. For each NABP, results are shown for two background configurations (input × 0.5 and input × 2). To evaluate this association, we analyzed ChIP-seq data for 32 DBPs in HepG2 cells and eCLIP data for 29 RBPs in K562 cells—two cell lines with the most comprehensive publicly available NABP binding site datasets from ENCODE and related studies. A gene was considered associated with an NABP if a ChIP-seq peak was located within ±5 kb of its transcription start site (TSS) or if an eCLIP peak was mapped to its transcribed region. As illustrated in Fig. 3a and Supplementary Fig. S1 , NABPs with experimentally validated binding sites consistently showed higher absolute contribution score values across their predicted target genes, reinforcing their regulatory relevance and corresponding with prior interpretations of DeepLIFT scores 25 , 26 . Model-derived importance scores reliably predict DBP binding sites, as evidenced by significant overlap with ChIP-seq peaks ( Fig. 3b , Supplementary Table S2 and S3, Supplementary Fig. S3a ). To evaluate this correspondence, we analyzed ChIP-seq data for 32 DNA-binding proteins (DBPs) and quantified how frequently high-scoring gene regions—defined by DeepLIFT contribution scores—coincided with experimental binding evidence. Overlap ratios were calculated as the proportion of high-contribution regions that intersected with ChIP-seq peaks, considering both promoter-proximal (±5 kb) and distal (±100 kb) regions around transcription start sites (TSSs). To account for the functional heterogeneity and abundance of ChIP-seq peaks, we applied a precision-at-K approach, focusing on the top-ranked or top-scoring gene regions most likely to reflect regulatory relevance. Four thresholds were used to define high-contribution regions: two based on contribution score rank (top/bottom 20 or 50) and two based on absolute score magnitude (≥ 0.01 or ≤ –0.01; ≥ 0.001 or ≤ –0.001). A control analysis using randomly selected gene regions of matched size was conducted in parallel. In all tested conditions, the overlap between ChIP-seq peaks and high-contribution regions exceeded that of the random control, supporting the model’s ability to prioritize functionally relevant regulatory sites. Similarly, as illustrated in Fig. 3c , Supplementary Table S4 , and Supplementary Fig. S3b , the model-derived importance scores consistently prioritize biologically relevant RBP–target interactions, as evidenced by the cumulative overlap between high-contribution gene regions and eCLIP peaks, which was substantially greater than that observed in the random control. We assessed this by analyzing eCLIP data for 29 RNA-binding proteins (RBPs) in K562 cells, comparing high-contribution regions—defined using the same thresholds as for DBPs—to eCLIP peaks mapped to transcribed regions (from TSS to TES). Collectively, these results demonstrate that the model effectively captures functionally meaningful NABP–target associations for both DNA- and RNA-binding proteins. Significant differences in the predicted regulatory targets of several NABPs were observed between K562 and HepG2 cells based on DeepLIFT contribution scores ( Fig. 3d ). To assess cell type–specific regulatory activity, we compared the DeepLIFT scores of genes harboring NABP binding sites, as defined by ChIP-seq or eCLIP. Because our deep learning model identifies target candidates from NABP-associated co-expressed gene sets, not all predicted targets necessarily reflect direct binding; rather, they represent genes that participate in shared or downstream regulatory programs. RPLP0 exhibited nearly identical score distributions across the two cell types, whereas ETF1, NELFE, and CAMTA2 showed marked divergence. These differences indicate that the model captures and adjusts NABP-dependent regulatory relationships in a cell type–specific manner. Deep learning uncovers functionally enriched NABP target genes beyond co-expression networks Deep learning predictions revealed that nucleic acid–binding proteins (NABPs) possess regulatory target genes with coherent and biologically specific functional enrichment, extending well beyond patterns detectable from co-expression alone ( Tables 1 , 2 and Supplementary Note 2 ). Across twelve representative NABPs 27 , 28 , predicted target sets consistently showed significant enrichment for biologically related Gene Ontology (GO) terms, whereas randomly sampled gene sets of matched size displayed no such enrichment ( Fig. 4 ). Notably, several enriched annotations were absent from the broader pools of co-expressed genes (Pearson’s z -score > 2), indicating that the model selectively extracts biologically relevant signals from large co-expression neighborhoods. Functional enrichment across GO Slim, Reactome, and PANTHER Pathways further confirmed this specificity ( Table 1 ) 29 . A ChatGPT-based gene set interpretation method independently supported these results, providing functional summaries and literature consistent with GO enrichment and UniProt annotations 30 ( Table 2 and Supplementary Note 2 ) 31 . Download figure Open in new tab Figure 4. Overrepresented functional annotation terms among target genes of nucleic acid-binding proteins. Panels a and b show results filtered at false discovery rate (FDR)–adjusted p -values of < 0.001 and 2, and (3) randomly selected genes of equal size. Gene Ontology (GO) terms from the categories of biological process, molecular function, and cellular component were combined. Fisher’s exact test p -values were adjusted using the Benjamini–Hochberg method to control the false discovery rate (FDR). View this table: View inline View popup Download powerpoint Table 1. Functional roles inferred for nucleic acid-binding proteins: integration of UniProt annotations, ChatGPT-based predictions, and confidence scores. Functional annotations obtained from UniProt entries were systematically compared with significantly enriched biological processes among the model-predicted regulatory targets, as identified by the PANTHER Overrepresentation Test (see Supplementary Table S5 ). Shared or overlapping functions between each nucleic acid-binding protein (NABP) and its predicted targets are indicated. Functional roles inferred using a ChatGPT-based interpretation method are highlighted in yellow, with confidence scores—defined as the proportion of annotated genes within the gene set—provided in parentheses. Confidence levels are categorized as high (0.87–1.00), medium (0.82–0.86), low (0.01–0.81), or none (0). For methodological reference, see Hu M. et al., Nat. Methods , 2025 ( https://doi.org/10.1038/s41592-024-02525-x ) 31 . View this table: View inline View popup Download powerpoint Table 2. Functional characterization of predicted regulatory targets of nucleic acid-binding proteins using a ChatGPT-based inference framework. Representative biological functions were inferred for each NABP by applying a large language model–based gene set interpretation method. Functional themes are summarized in bold for each NABP, while numbered subcategories denote distinct functional roles associated with individual gene subsets. Cell type–dependent functional specificity revealed by deep learning–refined NABP targets Deep learning refinement further exposed cell type–specific biological functions associated with regulatory targets of several RBPs. For example, the predicted targets of PKM showed distinct pathway enrichments between K562 cancer cells and neural progenitor cells (NPCs). Pathways such as Pentose phosphate pathway , Glycolysis , and Hypoxia response mediated by HIF activation were significantly enriched in K562 but not in NPCs, whereas Integrin signaling pathway , Glycogen metabolism , and Glutamate neurotransmitter release cycle were specifically enriched in NPCs using a background defined as 0.5× the original input feature values 32 – 41 ( Fig. 5 , Table 3 , Supplementary Table S5 ). ChatGPT-based interpretation produced consistent functional assignments ( Supplementary Note 2 ). Notably, these cell type–dependent associations align with known biology in cancer cells and NPCs. Download figure Open in new tab Figure 5. Functional mapping of nucleic acid-binding proteins reveals context-dependent regulatory roles in cancer and neural progenitor cells. A deep learning model trained on transcriptomic co-expression data was used to predict the regulatory influence of nucleic acid-binding proteins (NABPs) on gene expression. Contribution scores were computed to quantify the impact of each NABP across target genes, and the highest-ranking genes were designated as predicted regulatory targets. Functional interpretation was performed using a large language model–based inference method, which identified key biological pathways associated with each NABP. View this table: View inline View popup Download powerpoint Table 3. Functional enrichments among predicted regulatory target genes of nucleic acid-binding proteins. Functional enrichment analysis was performed using the PANTHER database. This table provides selected examples of annotations that were statistically enriched among regulatory target genes predicted by the deep learning model trained on gene co-expression data. ‘ Reference ’ indicates the total number of human genes; ‘ Analyzed ’ denotes the number of regulatory target genes examined; ‘ Expected ’ represents the number of genes expected to have the annotation by chance; ‘ Fold Enrichment ’ reflects the observed-to-expected ratio; ‘ Raw p -value ’ reports the result of Fisher’s exact test; and ‘ FDR ’ refers to the Benjamini–Hochberg false discovery rate. Additional RBPs exhibited similar specificity. Predicted target genes of ILF2 were enriched for G2/M DNA replication checkpoint in K562 cells but not in NPCs 42 – 44 ; targets of EIF2S2 were associated with Serine and glycine biosynthesis in NPCs but not in K562 45 , 46 ; and targets of DNAJC2 were enriched for the Ubiquitin proteasome pathway in K562 but not in NPCs 47 , 48 ( Fig. 5 , Table 3 , Supplementary Table S5 ). ChatGPT-based inference again recapitulated these cell type–dependent functional patterns ( Supplementary Note 2 ), consistent with previous reports. Together, these results demonstrate that the deep learning–based refinement of NABP target predictions enables systematic detection of cell type–dependent functions, revealing regulatory specificity not captured by co-expression alone. Functional enrichment analysis of NABP target genes using contribution score–based GSEA Gene set enrichment analysis (GSEA) based on DeepLIFT-derived contribution scores revealed robust, cell type–dependent functional redistribution among NABP regulatory targets. Rather than ranking genes by expression levels, we ranked predicted targets according to their relative regulatory contributions within each cell type, enabling pathway-level interpretation based on shifts in the positional distribution of functionally related genes along contribution-ranked lists, independent of absolute score magnitude or sign ( Fig 6a ). Using this framework, PKM-associated pathways exhibited pronounced redistribution between neural progenitor cells (NPCs) and K562 leukemia cells. Multiple Reactome pathways showed large differences in normalized enrichment scores (ΔNES = NES_K562 − NES_NPC), indicating substantial divergence in inferred regulatory coupling across cellular contexts. Pathways related to synaptic organization and neuronal system functions were enriched among positive ΔNES values (K562-biased), whereas immune-related pathways were preferentially represented among negative ΔNES values (NPC-biased). These contrasts were consistently captured across pathways ranked by ΔNES ( Fig. 6b , Table 4 , and Supplementary Fig. S5 ). Importantly, ΔNES reflects redistribution of contribution-ranked pathway positioning rather than direct activation or repression of lineage programs. Conceptually, ΔNES quantifies the relative positional shift of pathway-associated genes along contribution-ranked regulatory landscapes between cellular contexts. Download figure Open in new tab Download figure Open in new tab Figure 6. Pathway redistribution reveals a shared signaling backbone and context-dependent functional modules across RNA-binding proteins. a. Conceptual framework of ΔNES-based pathway analysis. Gene-level contribution scores were computed using DeepLIFT and used to rank genes for each RNA-binding protein (RBP) and cellular context. Gene set enrichment analysis (GSEA) was applied to Reactome pathways, and pathway redistribution between cell types was quantified as ΔNES (NES_K562 − NES_NPC), reflecting shifts in the positional distribution of pathway-associated genes along contribution-ranked lists. b. Representative GSEA enrichment plots illustrating pathway redistribution. Positive ΔNES values indicate relative enrichment in K562 cells, whereas negative ΔNES values indicate relative enrichment in neural progenitor cells (NPCs). These opposing patterns demonstrate redistribution of pathway-associated genes across contribution-ranked gene lists rather than direct pathway activation or repression. Other examples are shown in Supplementary Fig. S5 . c. Functional module–level organization of ΔNES-ranked pathways across RBPs (PKM, HNRNPK, and NELFE). A conserved signaling backbone is observed across all RBPs, alongside context-dependent redistribution of functional modules. Top (ΔNES > 0) and bottom (ΔNES < 0) pathways were grouped into functional modules based on shared biological processes. Numbers in the heatmap indicate the number of Reactome pathways assigned to each functional module. Color scale indicates pathway counts or enrichment distribution. Across all RBPs, broadly utilized signaling pathways—including receptor tyrosine kinase (RTK), MAPK, and WNT-related signaling—form a conserved signaling backbone shared across cellular contexts. In contrast, modules associated with neuronal functions (e.g., synaptic signaling) and immune processes are differentially enriched, indicating context-dependent functional redistribution. These results support a hierarchical model in which RBPs regulate common signaling networks while modulating cell type–specific functional outputs through redistribution of pathway-associated genes. View this table: View inline View popup Download powerpoint Table 4. Cell type–dependent enrichment of functional pathways identified by GSEA NES denotes the normalized enrichment score obtained from gene set enrichment analysis (GSEA). ΔNES represents the difference in NES between K562 and NPC (ΔNES = NES K 562 − NES NPC ). This table summarizes pathway enrichment results for PKM, shown as a representative RNA-binding protein consistent with Fig. 6 . The twenty Reactome pathways with the largest positive and negative ΔNES values are listed in the upper and lower sections, respectively. Pathways with large positive ΔNES values were predominantly annotated to high-level Reactome categories related to neural development and differentiation, whereas pathways with large negative ΔNES values were enriched for categories associated with immune regulation, cell cycle control, and genome maintenance, reflecting contrasting functional associations between NPCs and K562 cells. Corresponding analyses performed using an amplified DeepLIFT background (input × 2) yielded qualitatively consistent pathway-level contrasts, with generally higher enrichment scores due to an increased number of predicted PKM-associated target genes ( Supplementary Table S8 and S9 ). Because DeepLIFT contribution scores depend on background definition, we evaluated alternative reference configurations (input × 0.5 and input × 2). Although background scaling altered the polarity of contribution–expression correlations and the absolute direction of individual NES values, relative ΔNES-based pathway differences between NPC and K562 were preserved across configurations ( Fig. 6 ; Supplementary Tables S7–S9 ). The amplified background (input × 2) increased the number of genes with non-negligible contribution scores and generally yielded higher NES magnitudes, reflecting increased sensitivity without altering the relative redistribution patterns. These observations support the robustness of ΔNES as a comparative metric driven by stable rank-structure differences rather than background-dependent score polarity. Extending this analysis beyond PKM, similar category-level shifts were observed for additional RNA-binding proteins, including HNRNPK and NELFE ( Supplementary Tables S10–S13 ). Across all three proteins, Neural System–annotated pathways were significantly overrepresented among high ΔNES terms, whereas Immune System–annotated pathways were enriched among low ΔNES terms (one-sided Fisher’s exact test, p < 0.001; Supplementary Table S14 ). This reproducible enrichment pattern indicates systematic redistribution of regulatory influence between NPC and K562 contexts that is not restricted to a single protein, and remains consistent when evaluated at both pathway and functional module levels. Notably, Signal Transduction emerged as the dominant top-level Reactome category across the ΔNES spectrum for all three RBPs ( Supplementary Table S14 ). Subpathway analysis revealed that this enrichment was driven primarily by broadly utilized regulatory modules, including FGFR/RTK signaling, phospholipase C–mediated cascades, WNT/BMP pathways, and receptor-associated second-messenger systems. At the level of exact Reactome pathway identity, overlap across RBPs was limited, with only a single pathway (FGFR1C ligand binding and activation) shared among PKM, HNRNPK, and NELFE, and a small number of pairwise overlaps such as NF-κB activation and signaling between PKM and NELFE ( Supplementary Table S16 ). Consistent with this observation, pairwise comparison using the Jaccard similarity index revealed strong similarity between HNRNPK and NELFE (J ≈ 0.65), whereas PKM shared only limited overlap with either protein (PKM–HNRNPK: J ≈ 0.13; PKM–NELFE: J ≈ 0.15), indicating divergence at the level of individual pathway identity. However, grouping pathways into functional modules revealed a markedly different structure. Despite limited exact overlap, all three RBPs converged on common receptor-proximal signaling modules, particularly within the FGFR signaling axis and associated phospholipase C cascades, as well as ligand–receptor and MAPK-related signaling processes. These findings indicate that RBPs share a conserved signaling backbone at the functional level, even when individual pathway identities differ. Furthermore, PKM exhibited a broader and more diversified signaling profile, incorporating additional modules related to Rho GTPase signaling, Notch signaling, and neuronal-associated pathways, whereas bottom-ranked pathways were largely restricted to general signaling processes such as GPCR, WNT/BMP, and ligand-binding receptor pathways, consistent with leukemia/hematopoietic signaling states in K562 cells. Together, these results support a hierarchical model in which a shared signaling backbone is differentially extended by cell type–biased functional modules, rather than being defined by exact pathway overlap. To enable systematic interpretation, ΔNES-ranked pathways were grouped into functional modules based on shared biological processes (see Methods and Supplementary Table S15 ). This module-level organization revealed a clear separation between pathways with positive ΔNES values (relatively enriched in K562) and those with negative ΔNES values (relatively enriched in NPC) ( Fig. 6c ). Pathways with positive ΔNES were predominantly associated with signaling and interaction-related modules, including neuronal system–related signaling, metabolic processes, and stemness-associated programs. In contrast, pathways with negative ΔNES were enriched for immune system–related modules and developmental programs, such as organogenesis and lineage specification. Despite this divergence, a subset of signaling pathways—including receptor tyrosine kinase (RTK), MAPK, and WNT-related pathways—was consistently observed across RBPs, forming a shared signaling backbone. These findings indicate that RBPs do not primarily regulate distinct sets of cell-type-specific genes, but instead modulate common signaling networks while redistributing functional outputs across cellular contexts. Consistent with established roles of PKM2 in metabolic rewiring, nuclear transcriptional regulation, and proliferative state maintenance 49 , 50 , the observed ΔNES redistribution is compatible with context-dependent repositioning of metabolic and neuronal regulatory modules between leukemia and neural progenitor states 36 . HNRNPK and NELFE, which coordinate transcriptional and post-transcriptional regulatory processes during development and malignancy, displayed parallel redistribution patterns, further supporting a shared regulatory architecture modulated across cellular contexts ( Supplementary Note : Functional interpretation of ΔNES-based pathway redistribution across PKM, HNRNPK, and NELFE). Together, these findings demonstrate that contribution score–based GSEA captures reproducible, category-level functional redistribution across cell types while revealing a conserved upstream signaling backbone shared among multiple RNA-binding proteins. This framework provides an interpretable and scalable strategy for dissecting context-dependent regulatory organization beyond expression-centric or threshold-dependent analyses. To further investigate the gene-level composition of enriched pathways, we examined leading-edge genes identified by GSEA for ΔNES-ranked Reactome pathways. Notably, many leading-edge genes were classified as shared functional genes rather than strictly cell-type-specific genes, even within pathways associated with neuronal or immune systems ( Supplementary Table S17 ). These results suggest that RBPs predominantly modulate broadly shared gene networks while contributing to context-dependent functional differences at the pathway level. Discussion Our findings demonstrate that RNA-binding proteins (RBPs) regulate gene expression programs through redistribution of regulatory influence across a shared signaling backbone, rather than by controlling entirely distinct pathway sets. Across multiple RBPs, including PKM, HNRNPK, and NELFE, Signal Transduction consistently forms a conserved signaling backbone, while functional modules associated with neuronal and immune processes are differentially positioned depending on cellular context. This hierarchical organization indicates that regulatory rewiring occurs through redistribution across common signaling networks. Importantly, ΔNES captures this redistribution as shifts in the positional distribution of pathway-associated genes along contribution-ranked lists, rather than reflecting direct pathway activation or repression. Because gene expression represents the combined effects of multiple regulatory inputs, contribution scores quantify relative regulatory influence, enabling detection of subtle but systematic differences in regulatory architecture across cell types. At the pathway level, the shared signaling backbone is characterized by broadly utilized receptor-mediated and growth factor–associated pathways, including RTK, MAPK, and WNT-related signaling. Although individual pathway identities vary across RBPs, functional module–level analysis reveals convergence on common upstream signaling processes. This structural convergence distinguishes our findings from conventional enrichment-based analyses, which typically emphasize pathway identity rather than organization. In contrast, modules associated with neuronal functions and immune responses exhibit context-dependent redistribution, indicating that RBPs modulate how shared gene networks are functionally deployed across cellular states. This model contrasts with transcription factors, which often determine cellular identity through selective activation of more restricted gene sets. Instead, RBPs appear to regulate functional allocation within broadly shared signaling and metabolic networks. The robustness of these patterns across alternative DeepLIFT background settings further supports the interpretation that pathway-level contrasts arise from stable rank-structure differences rather than background-dependent artifacts. While co-expression–based inference does not directly establish causal regulatory relationships, it captures functional coupling at the systems level, providing a scalable framework for analyzing regulatory architecture in the absence of experimentally defined binding information. Looking forward, expanding this framework to a broader range of cell types and regulatory factors will enable systematic characterization of pathway redistribution across diverse biological contexts. Comparative analysis across transcription factors and RNA-binding proteins may further clarify their distinct and complementary roles in shaping regulatory architectures. Extending the framework to inter-individual variation and disease-associated transcriptomes, as demonstrated in related deep learning studies of gene expression regulation, may provide insight into the regulatory basis of phenotypic variation and disease susceptibility. Such analyses could facilitate the identification of early disease-associated shifts in regulatory architecture and support applications in diagnosis and preventive medicine. Furthermore, by linking predicted regulatory targets to pathway-level redistribution patterns, this framework may contribute to the reconstruction of gene regulatory networks and enable in silico prediction of transcriptional responses to perturbations. These capabilities could support the identification of candidate therapeutic targets and guide strategies for modulating gene expression programs in a controlled manner. Together, our results establish contribution score–based pathway redistribution as a framework for uncovering conserved signaling architectures alongside context-dependent functional modules, providing a systems-level perspective on gene regulation beyond expression-centric analyses. Methods Predictive modeling of gene expression across diverse cell types using deep learning and NABP binding information To predict gene expression across multiple human cell types, we adapted a previously established deep learning framework that integrates DNA-binding site data for transcription factors and other regulatory proteins. In this implementation, regulatory elements located both proximally and distally (up to ±1 Mb) from transcription start sites (TSSs) were included as input features ( Fig. 1 ) 25 , 51 . To minimize the effects of cellular heterogeneity in bulk tissue, RNA-seq datasets were selected at the individual cell-type level. Publicly available RNA-seq data were retrieved for four functionally distinct cell types: human foreskin fibroblasts (HFFs), mammary epithelial cells (HMECs), neural progenitor cells (NPCs), and either HepG2 hepatoma cells or K562 leukemia cells, depending on the class of regulatory protein analyzed 52 , 53 . Gene expression values (FPKM) were computed using the RSeQC toolkit based on GENCODE v19 annotations, and subsequently log₂-transformed after adding a small offset to avoid undefined values. Following the approach described by Tasaki et al. 25 , 51 , expression fold changes were calculated relative to the median expression level across the four cell types. For each model iteration, three common cell types (HFFs, HMECs, and NPCs) were used, and either HepG2 (for DNA-binding protein analysis) or K562 (for RNA-binding protein analysis) was included to match the availability of ChIP-seq and eCLIP validation datasets. These cell types were chosen to represent diverse biological lineages and regulatory contexts. For specific analyses requiring direct comparison of DeepLIFT contribution scores across cell types—such as the evaluation of cell type–dependent regulatory differences for the same nucleic acid-binding protein (e.g., Fig. 3d )—models were trained using a four-cell-type configuration comprising HFFs, NPCs, HepG2, and K562. This unified configuration enabled direct comparison of contribution score distributions between HepG2 and K562 while maintaining a shared background of common cell types. Extended regulatory features and DeepLIFT-compatible model design To capture distal regulatory signals beyond promoter regions, we integrated DNA-binding site data from putative enhancers into a deep learning framework for gene expression prediction. Building upon our earlier work 25 , 51 , this model extends the architecture originally proposed by Tasaki et al. (2020) 25 , 51 , which focused on promoter-proximal regions and included RNA-binding protein and miRNA inputs. In our previous study, aimed at identifying insulator-associated DNA-binding proteins, we deliberately excluded RNA-binding proteins and miRNAs to focus on DNA-centered regulatory elements. Instead, we combined DNA-binding sites from both promoter (−2 kb to +1 kb) and enhancer-like regions (±1 Mb from the transcription start site [TSS]) into a unified input matrix, enabling DeepLIFT analysis under the high-accuracy “genomics-default” mode ( Supplementary Fig. S6 ). In this study, we use DeepLIFT to compute attribution values for each gene. We refer to these values as DeepLIFT scores, also described as importance scores or contribution scores in the context of feature attribution. Binding site coordinates were obtained from the GTRD database (v20.06) 54 , and regulatory associations were inferred using cis-eQTLs from GTEx v8. A ChIP-seq peak that overlapped a 100-bp window centered on an expression-associated SNP was considered linked to the corresponding transcript. Trans-eQTLs were excluded, and coordinates were converted to the hg19 assembly using UCSC LiftOver 55 . To limit feature dimensionality while maximizing genomic coverage, we computed peak lengths for each DNA-binding protein (DBP) in 100-bp bins across ±1 Mb around the TSS, then retained the maximum peak value within each 10-kb window, yielding up to 200 enhancer bins. These were concatenated with 30 promoter-region bins to form a 230-bin unified input layer. This single-input-layer structure was necessary because DeepLIFT’s genomics-default configuration (“RevealCancel rule on dense layers and Rescale rule on conv layers”) does not support models with multiple input layers. Concatenating all DBP-related features into one matrix thus maintained compatibility while enhancing interpretability. Model training followed the protocol described by Tasaki et al. and applied in our prior work 25 , 51 . Specifically, 2,705 genes from chromosome 1 were used for testing, while 22,251 genes from other chromosomes were randomly assigned to the training set and 2,472 to the validation set. Models were implemented in TensorFlow 2.6.0 and trained on an NVIDIA RTX 3080 GPU, typically completing in 2 to 4 hours. Feature importance was assessed using DeepLIFT, which assigned contribution scores to input features indicating their relative influence on gene expression prediction. Predicting NABP regulatory targets using gene co-expression–informed substitutions To uncover regulatory targets of NABPs without relying on direct binding site assays, we implemented a co-expression–driven modification of the deep learning input. Specifically, DNA-binding proteins with low contribution scores in the original model were replaced with NABPs whose regulatory interactions were inferred from gene co-expression data. Of the 1,310 DBPs initially included, 302 were substituted with other DBPs, and in a separate RNA-binding protein analysis, 325 were replaced with RNA-binding proteins (RBPs) identified from the ENPD database 24 . Putative NABP–target relationships were established using co-expression profiles provided by COXPRESdb 15 , where a Pearson’s z -score > 2 between a NABP-encoding gene and a non-NABP gene indicated a strong co-expression link. These co-expression pairs served as proxies for regulatory interactions and were used to generate new input features for the model. To integrate these proxies into the deep learning framework, hypothetical NABP binding sites were assigned at the transcription start site (TSS) and 2 kb upstream of each co-expressed gene. The model was then retrained using this modified input, and DeepLIFT was applied to calculate feature-level importance scores (hereafter referred to as contribution scores), which quantify the influence of each NABP–gene interaction on the model’s predicted expression output. These contribution scores represent feature-wise attributions derived from the trained deep learning model and do not correspond directly to measured gene expression levels, which reflect the integrated effects of multiple regulatory inputs. This approach enables the functional characterization of NABPs with unknown or uncharacterized binding motifs by leveraging gene co-expression as a proxy for regulatory targeting. Validation of predicted NABP targets using experimentally determined binding To assess the biological relevance of the predicted regulatory targets, we compared deep learning–inferred NABP–gene associations with experimentally validated binding data. Specifically, we evaluated the spatial overlap between predicted target genes and ChIP-seq peaks for 32 DNA-binding proteins (within ±5 kb of transcription start sites [TSSs]) as well as eCLIP peaks for 29 RNA-binding proteins (within the transcribed region from TSS to transcription end site [TES]). Gene structure annotations were derived from the GENCODE v19 release 56 , and binding site data were obtained from the ENCODE portal and prior publications ( Supplementary Table S6 ) 57 . This overlap analysis provides orthogonal validation for the predicted targets and supports the utility of gene co-expression–based modeling for inferring NABP–target relationships. Prioritization of NABP regulatory targets based on importance score rankings To identify the primary regulatory targets of each NABP, genes were ranked based on the importance scores assigned to NABP binding sites by the deep learning model. Genes with binding sites consistently ranked in the top five or bottom five across all NABPs were selected as putative targets, reflecting high positive or negative regulatory influence. These targets represent genes most strongly associated with a given NABP’s regulatory activity. To ensure sufficient statistical power for downstream enrichment analysis, rank thresholds were adjusted for each NABP based on the number of high-confidence predictions. While a contribution score rank cutoff of five was sufficient in most cases, the threshold was relaxed in specific cases—for example, up to rank ten for proteins, which initially yielded a smaller number of predicted targets. Functional enrichment and AI-based interpretation confirm the biological relevance of NABP regulatory targets. To evaluate the biological coherence of predicted NABP regulatory targets, we performed functional enrichment analysis using the Statistical Overrepresentation Test from the PANTHER database alongside custom in-house scripts 29 . Gene Ontology (GO) terms from biological process, molecular function, and cellular component categories were analyzed for enrichment across three gene sets: (1) primary regulatory target genes identified by the model, (2) all co-expressed genes (Pearson’s z -score > 2), and (3) randomly selected genes of matched size. Enrichment was assessed using Fisher’s exact test with Benjamini–Hochberg false discovery rate (FDR) correction, and FDR < 0.05 was considered statistically significant. For this analysis, human gene symbols were mapped to Ensembl Gene IDs using the HGNC database ( https://www.genenames.org/ ), and GO annotations were obtained from the Gene Ontology resource 27 , 28 . Target gene IDs served as the test set, while the full expression dataset used in the deep learning model formed the reference background. Gene set enrichment analysis (GSEA) was performed using the pre-ranked GSEA algorithm implemented in GSEApy (v1.1.11). For each DNA-binding protein (DBP) and RNA-binding protein (RBP), DeepLIFT contribution scores were obtained for all genes from the trained deep learning models. Because raw DeepLIFT scores vary in magnitude across genes and include a large number of zeros, scores were normalized within each gene by scaling the values across all 1,310 DBPs/RBPs to a continuous range from −1 to 1 using min–max normalization. Genes were then ranked according to these normalized contribution values specific to the DBP or RBP of interest. The ranked lists were used as input for GSEA with the Reactome (v2025.1) and Gene Ontology (Molecular Function, Biological Process, and Cellular Component, 2021 releases) gene set collections. Analyses were performed with 10,000 permutations, weighted enrichment statistics (weight = 1), and a minimum gene-set size of 1 to retain all functional categories. Enrichment significance was assessed using normalized enrichment scores (NES), nominal p -values, and false-discovery rates (FDR). This procedure enabled the identification of pathways enriched among genes with high positive or negative DeepLIFT contributions, thereby revealing regulatory roles and cell type–specific pathways associated with each nucleic acid–binding protein. Pairwise similarity between Signal Transduction subpathway sets was quantified using the Jaccard similarity index, defined as the size of the intersection divided by the size of the union of the compared sets. Unlike conventional pathway activity methods such as GSVA 58 or AUCell 59 , which rank genes based on expression levels, our framework ranks genes using deep learning–derived contribution scores, enabling pathway analysis in terms of inferred regulatory influence. To complement the statistical analyses, we applied a ChatGPT-based functional inference method 31 . This approach used curated prompts—detailed in Extended Data Tables 1 and 5 —to predict functional roles based on gene sets, and retrieved relevant literature using a modified version of the “4.Reference search and validation.ipynb” script 31 . This script is publicly available on Code Ocean and GitHub ( https://doi.org/10.24433/CO.7045777.v1 ; https://github.com/idekerlab/llm_evaluation_for_gene_set_interpretation ). The ChatGPT-based results provided consistent and interpretable biological themes aligned with the enrichment findings. Definition of functional modules for Reactome pathways To facilitate interpretation of pathway-level results, individual Reactome pathways were grouped into higher-order functional modules based on shared biological processes. Functional modules were manually curated by integrating pathway annotations, pathway names, and known biological relationships among pathways. Pathways belonging to the same top-level Reactome category were first examined and further subdivided into functionally coherent modules (e.g., synaptic signaling, receptor tyrosine kinase signaling, immune receptor signaling, and metabolic sub-pathways). This grouping was performed to reduce redundancy among closely related pathways and to enable more interpretable visualization of pathway enrichment patterns. Each pathway was assigned to a single functional module, and a corresponding abbreviated module name (“module short”) was defined for compact representation in figures. The complete mapping between Reactome pathways and functional modules is provided in Supplementary Table S15 . Although this module definition involves manual curation, the grouping was guided by established biological knowledge and Reactome pathway structure, ensuring consistency across categories and RBPs. Statistical analysis Functional enrichment was assessed using Fisher’s exact test, with multiple testing correction applied via the Benjamini–Hochberg false discovery rate (FDR) procedure. Statistical significance was defined as FDR < 0.05. Distributions of DeepLIFT contribution scores for nucleic acid-binding sites derived from gene co-expression data and ChIP-seq data were compared using the Mann–Whitney U test (also known as the Wilcoxon rank-sum test). A p -value < 0.05 was considered statistically significant. GSEA (gene set enrichment analysis) was performed using DeepLIFT contribution scores of DBPs and RBPs. The normalized enrichment score (NES) was used to quantify whether regulatory target genes associated with specific biological functions were preferentially located at the high or low end of the ranked gene list. Statistical significance was assessed using permutation-based p -values, as implemented in the GSEA framework. Data availability RNA-seq data for human foreskin fibroblast (HFF) cells were obtained from the Gene Expression Omnibus (GEO) under accession number GSM941744 . RNA-seq data for human mammary epithelial cells (HMEC) were retrieved from ENCODE via the UCSC Genome Browser ( http://genome.ucsc.edu/encode/ ) under the file name wgEncodeCshlLongRnaSeqHmecCellPapAlnRep1.bam . RNA-seq data for human neural progenitor cells (NPC) were obtained from GEO under accession number GSM915326 . RNA-seq data for HepG2 and K562 cells were obtained from the ENCODE portal under accession numbers ENCFF670LIE and ENCFF584WML , respectively. mRNA co-expression data were downloaded from COXPRESdb v7 ( https://zenodo.org/record/6861444 ). The list of nucleic acid-binding proteins was obtained from the Eukaryotic Nucleic acid Binding Protein (ENPD) database ( http://qinlab.sls.cuhk.edu.hk/ENPD/ ). ChIP-seq data for transcription factors were downloaded from the GTRD database (v20.06 release; https://gtrd.biouml.org ). eQTL data ( GTEx_Analysis_v8_eQTL.tar ) were obtained from the GTEx portal (v8 release; https://gtexportal.org ). eCLIP data for RNA-binding proteins were retrieved from the ENCODE portal (accession ENCSR456FVU ) and from Supplementary Data 2 of Van Nostrand et al., Nature (2020), https://doi.org/10.1038/s41586-020-2077-3 . Code availability The deep learning model implemented in this study is available at https://github.com/reposit2/coexgene . Author contributions N.O. designed and conducted the research, and wrote and revised the manuscript. K.S. revised the manuscript. Supplementary data Supplementary data are available at Conflict of interests None declared. Funding This work was supported by the Japan Society for the Promotion of Science (JSPS) KAKENHI [Grant Number 16K00387], a start-up research grant from Osaka University, the KIOXIA Young Scientist Research Grant at Waseda University, the Grant for Basic Science Research Projects from the Sumitomo Foundation, the Research Grant from the CASIO Science Promotion Foundation, and the Research Grant from the Okawa Foundation for Information and Telecommunications awarded to N.O., JST CREST [Grant Number JPMJCR23N1], and the GteX Program Japan [Grant Number JPMJGX23B4] awarded to K.S. Supplementary Note and Tables Supplementary Note for Figure 3 . Definition of contribution score–based regions, thresholds, and enrichment analyses Contribution score ranking and binning For each nucleic acid-binding protein (NABP), DeepLIFT contribution scores were computed for all predicted gene-level regulatory regions and ranked from highest positive to lowest negative values. Ranked scores were divided into consecutive bins of 50 ranks, yielding 26 bins per NABP (rank 1: 1–50; rank 2: 51–100; …; rank 26: 1,251–1,310), unless otherwise specified. Definition of high- and low-contribution regions High- and low-contribution regions were defined using both rank-based and score-based thresholds to assess the robustness of enrichment results. Rank-based thresholds included upper and lower rank cutoffs (e.g., top or bottom 20 and 50 ranks), while score-based thresholds were defined using absolute contribution score cutoffs (e.g., ≥ 0.01 or ≥ 0.001). All threshold combinations used in the analyses are summarized in Tables S2 and S3. Genomic regions used for overlap analyses For DNA-binding proteins, ChIP-seq overlap analyses were conducted within two genomic contexts: promoter-only regions defined as ±5 kb from the transcription start site (TSS), and extended promoter–enhancer regions defined as ±100 kb from the TSS. For RNA-binding proteins, eCLIP overlap analyses were restricted to transcribed regions spanning from the TSS to the transcription end site (TES), including both exonic and intronic intervals. Random controls Random control regions were generated by sampling genomic intervals matched for size and genomic context, excluding annotated binding sites, to estimate the expected overlap by chance. All enrichment and overlap analyses were repeated using these controls for comparison. Background configuration for DeepLIFT analyses Unless otherwise noted, DeepLIFT contribution scores were computed using a background defined as 0.5× the original input feature values. Analyses using an alternative background configuration (input × 2) yielded qualitatively similar results and are presented in Supplementary Figures S2 and S3. Cell type configurations used for model training and comparative analyses To accommodate differences in experimental data availability and analytical objectives, multiple cell type configurations were used across analyses. For the primary analyses of DNA-binding proteins (DBPs), models were trained using expression data from HFFs, HMECs, NPCs, and HepG2, enabling validation against ChIP-seq datasets available for HepG2 cells. For RNA-binding protein (RBP) analyses, models were trained using HFFs, HMECs, NPCs, and K562 to facilitate validation using eCLIP datasets available for K562 cells. For analyses requiring direct comparison of regulatory contribution patterns between HepG2 and K562 for the same nucleic acid-binding proteins (e.g., Fig. 3d ), a unified four–cell-type configuration comprising HFFs, NPCs, HepG2, and K562 was employed. This configuration enabled direct comparison of DeepLIFT contribution score distributions across these two cell types while maintaining a shared transcriptomic background. In all cases, cell type selection was driven by analytical requirements and the availability of independent experimental validation data, and no validation datasets were used as direct inputs during model training. Gene co-expression–based predictions align with ChIP-seq–defined DNA-binding targets High-ranking predicted binding regions frequently coincide with ChIP-seq peaks, particularly in promoter–enhancer intervals. Deep learning–derived DNA-binding site predictions based on gene co-expression data show substantial concordance with ChIP-seq results, especially when broader genomic regions encompassing promoters and putative enhancers are considered. When overlap was assessed within ±100 kb of transcription start sites (TSSs)—a window supported by prior enhancer–gene mapping studies 60 , 61 —the mean and median overlap ratios reached 0.70 and 0.77, respectively (Ratio A2 / B2; Supplementary Tables S2 and S3 ). In contrast, overlap was lower for promoter regions alone (±5 kb), with corresponding mean and median values of 0.35 and 0.34. Several DNA-binding proteins, including AKAP8, POGK, SPEN, THAP7, ZNF142, and ZNF691, showed notably high promoter-region overlap ratios (>0.5), suggesting potential direct regulatory activity. Nevertheless, not all ChIP-seq peaks necessarily reflect functionally relevant transcriptional binding events. For RNA-binding proteins, overlap was examined within the transcribed regions of RNA molecules, and the mean and median overlap ratios were 0.70 and 0.76, respectively (Ratio A2 / B2; Supplementary Table S4 ). These findings indicate that model-inferred high-scoring sites do not universally coincide with experimentally determined binding regions, underscoring the complexity and context dependence of regulatory interactions. Functional interpretation of ΔNES-based pathway redistribution across PKM, HNRNPK, and NELFE To contextualize the ΔNES-based redistribution patterns described in the main text, we examined established biological roles of PKM, HNRNPK, and NELFE in relation to the pathway categories identified by contribution score–based GSEA. PKM PKM2, the predominant isoform in proliferative and stem-like states, is widely recognized as a central mediator of oncogenic metabolic rewiring and has been shown to exert non-canonical nuclear functions that coordinate transcriptional programs during tumorigenesis 49 , 50 . Beyond its canonical glycolytic role, PKM2 regulates gene expression through protein–protein interactions with transcriptional regulators and chromatin-associated factors. Isoform switching between PKM2 and PKM1 accompanies metabolic transitions during differentiation, including neuronal maturation, where PKM1-enriched states are associated with oxidative metabolism and terminal differentiation 36 . PKM has also been identified as a ribosome-associated factor capable of modulating mRNA translation 62 , and accumulating evidence suggests that metabolic enzymes, including pyruvate kinase, can function as non-canonical RNA-binding proteins 63 . In addition to its roles in metabolism and transcriptional regulation, PKM2 has emerged as a key regulator of immunometabolic processes. PKM2-dependent metabolic reprogramming has been shown to modulate cytokine production, including both pro-inflammatory and anti-inflammatory responses, and to promote immune cell proliferation and activation 64 – 66 . These findings indicate that PKM2 links metabolic state to immune signaling outputs, consistent with the enrichment of immune-associated pathways observed in ΔNES-based analyses. PKM2 is also tightly integrated with major signaling pathways. For example, receptor tyrosine kinase signaling, including fibroblast growth factor receptor 1 (FGFR1), directly phosphorylates PKM2, inducing a shift from an active tetrameric form to a less active dimeric state and thereby promoting the Warburg effect 67 . In addition, PKM2 has been functionally connected to NOTCH signaling, where coordinated regulation of PKM2 and NOTCH1 influences tumor growth and metastasis 68 , as well as to GPCR-associated pathways, such as the GPR3–β-arrestin2–PKM2 axis, which regulates glycolysis and liver pathophysiology 69 . PKM2 further contributes to TGF-β signaling by stabilizing TGF-β receptor complexes, thereby promoting fibrotic responses 70 . Beyond these pathways, PKM2 participates in diverse context-dependent processes, including lipid metabolism, angiogenesis, and metastatic progression. PKM2 has been implicated as a regulator of lipid metabolic programs, particularly in cardiac tissues 71 , 72 , and promotes tumor angiogenesis by modulating endothelial cell junction dynamics and ATP production 73 , 74 . Oxidative modification of PKM2 has also been linked to enhanced metastatic potential in cancer cells 75 . Together, these multifunctional properties position PKM2 at the interface of metabolism, signaling, and gene regulation. This integrated regulatory capacity is consistent with the ΔNES-inferred redistribution of neuronal, immune, metabolic, and signaling-associated pathways across neural progenitor and leukemia contexts, supporting a model in which PKM2 contributes to context-dependent repositioning of shared functional networks rather than driving strictly lineage-specific pathway activation 76 – 78 . HNRNPK HNRNPK is a multifunctional RNA-binding protein that integrates transcriptional and post-transcriptional regulatory processes. Haploinsufficiency of HNRNPK predisposes to hematologic malignancies and disrupts proliferation and differentiation programs in myeloid lineages 79 . In stem and progenitor contexts, HNRNPK coordinates transcriptional activation of proliferation-associated genes, including MYC and growth factor signaling components, while simultaneously regulating RNA stability and processing 80 . These dual transcriptional and post-transcriptional roles support a model in which HNRNPK contributes to lineage specification and tumor-associated transcriptional remodeling. Accordingly, the redistribution of immune- and neural-associated pathway annotations across cellular contexts observed here is compatible with a broader function of HNRNPK in modulating differentiation–proliferation balance rather than directing a single lineage-specific program. NELFE NELFE is a subunit of the negative elongation factor (NELF) complex that regulates promoter-proximal RNA polymerase II pausing, a metazoan-specific mechanism critical for rapid transcriptional responsiveness and developmental gene regulation 81 , 82 . Oncogenic activation of NELFE has been shown to enhance MYC-associated transcriptional programs and promote tumor progression 83 . In hematopoietic differentiation, dynamic modulation of NELF protein abundance alters genome-wide transcriptional pausing and controls granulocytic lineage commitment 84 . Furthermore, NELF regulates progenitor expansion and stem cell pool maintenance in regenerative contexts through modulation of p53-dependent transcriptional programs 85 . Collectively, these findings indicate that NELFE functions as a context-dependent regulator of transcriptional architecture, influencing lineage transitions through control of elongation dynamics. The parallel ΔNES redistribution patterns observed for NELFE and HNRNPK are therefore consistent with convergence of distinct RNA-binding proteins on shared regulatory infrastructures governing differentiation and proliferative states. Shared signaling architecture Across all three proteins, Signal Transduction emerged as the dominant top-level Reactome category. Subpathway analysis revealed enrichment of broadly utilized regulatory modules, including FGFR/RTK signaling, IRS-mediated cascades, WNT/BMP pathways, and receptor-associated second-messenger systems. Although neuronal receptor–related pathways were present, they constituted a minority of enriched subpathways ( Fig. 6c ). This distribution argues against a simple reflection of neuron-specific transcriptional signatures and instead supports the presence of a conserved upstream signaling backbone that is differentially interpreted across cellular contexts. Importantly, the ΔNES-based framework does not imply direct activation or repression of lineage programs by individual proteins. Rather, it captures hierarchical shifts in the positioning of pathway-associated genes along contribution-ranked lists, reflecting redistribution of inferred regulatory influence across cellular states. The convergence of redistribution patterns across metabolically (PKM), transcriptionally integrative (HNRNPK), and elongation-regulatory (NELFE) proteins supports the broader interpretation that contribution-based pathway ranking can reveal conserved regulatory scaffolds underlying context-dependent functional organization. View this table: View inline View popup Download powerpoint Supplementary Table S1. Co-expression–based selection enhances the contribution scores of NABP binding sites. Binding sites identified through gene co-expression analysis exhibited higher absolute DeepLIFT contribution scores than those selected solely from ChIP-seq data without co-expression information. This trend was consistently observed for DNA-binding proteins and was significantly stronger for RNA-binding proteins across the analyzed cell types. For each group, the table presents the mean and median values of both positive and negative contribution scores. These results indicate that gene co-expression–based selection preferentially highlights functionally relevant regulatory interactions. The upper and lower sections of the table correspond to analyses performed using DeepLIFT background references scaled to input × 0.5 and input × 2, respectively. Supplementary Table S2. Enrichment of high-contribution promoter regions with experimentally validated DBP binding sites. (See Supplementary Excel file) This table summarizes the proportion of promoter regions (±5 kb from transcription start sites [TSSs]) that were predicted to be functionally important by DeepLIFT contribution scores and validated by ChIP-seq data for the corresponding DNA-binding proteins (DBPs). Overlap ratios were calculated across four thresholds—two based on rank and two based on absolute score—to assess the robustness of the predictions. High- and low-contribution regions were classified using the following thresholds for the 1,310 NABPs analyzed: Rank-based thresholds: (A1 and B1) proteins ranked in the top or bottom 20 (i.e., rank ≤ 20 or ≥ 1,290) (A2 and B2) proteins ranked in the top or bottom 50 (i.e., rank ≤ 50 or ≥ 1,260) Score-based thresholds : (C1 and D1) contribution score ≥ 0.01 or ≤ –0.01 (C2 and D2) contribution score ≥ 0.001 or ≤ –0.001 Supplementary Table S3. Enrichment of high-contribution promoter and enhancer regions with ChIP-seq–validated DBP binding sites. This table summarizes the proportion of predicted high-contribution regions—defined within ±100 kb of transcription start sites (TSSs), encompassing both promoter and distal enhancer regions—that overlap with ChIP-seq binding sites for the corresponding DNA-binding proteins (DBPs). Strong concordance between model-derived predictions and ChIP-seq peaks across multiple thresholds highlights the reliability of the contribution scoring method for identifying functionally relevant regulatory sites. High- and low-contribution regions were classified using the following thresholds for the 1,310 NABPs analyzed: Rank-based thresholds: (A1 and B1) proteins ranked in the top or bottom 20 (i.e., rank ≤ 20 or ≥ 1,290) (A2 and B2) proteins ranked in the top or bottom 50 (i.e., rank ≤ 50 or ≥ 1,260) Score-based thresholds : (C1 and D1) contribution score ≥ 0.01 or ≤ –0.01 (C2 and D2) contribution score ≥ 0.001 or ≤ –0.001 Supplementary Table S4. Enrichment of high-contribution transcribed regions with eCLIP–validated RBP binding sites. This table summarizes the proportion of predicted high-contribution regions—defined within the transcribed regions of genes (from transcription start site [TSS] to transcription end site [TES])—that overlap with eCLIP peaks for the corresponding RNA-binding proteins (RBPs). Strong concordance between model-based predictions and eCLIP peaks across multiple thresholds underscores the reliability of the contribution scoring method for identifying functionally relevant regulatory sites. High- and low-contribution regions were classified using the following thresholds applied to the 1,310 NABPs analyzed: Rank-based thresholds: (A1 and B1) proteins ranked in the top or bottom 20 (i.e., rank ≤ 20 or ≥ 1,290) (A2 and B2) proteins ranked in the top or bottom 50 (i.e., rank ≤ 50 or ≥ 1,260) Score-based thresholds : (C1 and D1) contribution score ≥ 0.01 or ≤ –0.01 (C2 and D2) contribution score ≥ 0.001 or ≤ –0.001 Acknowledgements Supercomputing resources were provided by the Human Genome Center, the Institute of Medical Science, the University of Tokyo, and the ROIS National Institute of Genetics. Funder Information Declared Japan Society for the Promotion of Science (JSPS) KAKENHI a start-up research grant from Osaka University KIOXIA Young Scientist Research Grant at Waseda University Grant for Basic Science Research Projects from the Sumitomo Foundation Japan Science and Technology Agency Research Grant from the CASIO Science Promotion Foundation Research Grant from the Okawa Foundation Japan Science and Technology Agency CREST Japan Science and Technology Agency GteX Footnotes revised the abstract, introduction, results, discussion, and supplemental materials https://github.com/reposit2/coexgene References ↵ Belz , G. T. & Nutt , S. L . Transcriptional programming of the dendritic cell network . Nature reviews. Immunology 12 , 101 ‒ 113 ( 2012 ). doi: 10.1038/nri3149 OpenUrl CrossRef PubMed ↵ Miller , J. C. et al. Deciphering the transcriptional network of the dendritic cell lineage . Nature immunology 13 , 888 ‒ 899 ( 2012 ). doi: 10.1038/ni.2370 OpenUrl CrossRef PubMed Ravasi , T. et al. An atlas of combinatorial transcriptional regulation in mouse and man . Cell 140 , 744 ‒ 752 ( 2010 ). doi: 10.1016/j.cell.2010.01.044 OpenUrl CrossRef PubMed Web of Science Gertz , J. et al. Distinct properties of cell-type-specific and shared transcription factor binding sites . Molecular cell 52 , 25 ‒ 36 ( 2013 ). doi: 10.1016/j.molcel.2013.08.037 OpenUrl CrossRef PubMed Web of Science Lambert , S. A. et al. The Human Transcription Factors . Cell 172 , 650 ‒ 665 ( 2018 ). doi: 10.1016/j.cell.2018.01.029 OpenUrl CrossRef PubMed Vaquerizas , J. M. , Kummerfeld , S. K. , Teichmann , S. A. & Luscombe , N. M . A census of human transcription factors: function, expression and evolution . Nature reviews. Genetics 10 , 252 ‒ 263 ( 2009 ). doi: 10.1038/nrg2538 OpenUrl CrossRef PubMed Web of Science Pierson , E. et al. Sharing and Specificity of Co-expression Networks across 35 Human Tissues . PLoS Comput Biol 11 , e1004220 ( 2015 ). doi: 10.1371/journal.pcbi.1004220 OpenUrl CrossRef PubMed Sonawane , A. R. et al. Understanding Tissue-Specific Gene Regulation . Cell Rep 21 , 1077 ‒ 1088 ( 2017 ). doi: 10.1016/j.celrep.2017.10.001 OpenUrl CrossRef PubMed Saha , A. et al. Co-expression networks reveal the tissue-specific regulation of transcription and splicing . Genome research 27 , 1843 ‒ 1858 ( 2017 ). doi: 10.1101/gr.216721.116 OpenUrl Abstract / FREE Full Text ↵ Boyle , A. P. et al. Comparative analysis of regulatory information and circuits across distant species . Nature 512 , 453 ‒ 456 ( 2014 ). doi: 10.1038/nature13668 OpenUrl CrossRef PubMed Web of Science ↵ Kribelbauer , J. F. , Rastogi , C. , Bussemaker , H. J. & Mann , R. S . Low-Affinity Binding Sites and the Transcription Factor Specificity Paradox in Eukaryotes . Annu Rev Cell Dev Biol 35 , 357 ‒ 379 ( 2019 ). doi: 10.1146/annurev-cellbio-100617-062719 OpenUrl CrossRef PubMed ↵ Ray , D. et al. RNA-binding proteins that lack canonical RNA-binding domains are rarely sequence-specific . Sci Rep 13 , 5238 ( 2023 ). doi: 10.1038/s41598-023-32245-9 OpenUrl CrossRef PubMed Zeke , A. et al. Deep structural insights into RNA-binding disordered protein regions . Wiley Interdiscip Rev RNA 13 , e1714 ( 2022 ). doi: 10.1002/wrna.1714 OpenUrl CrossRef PubMed ↵ Zaharias , S. et al. Intrinsically disordered electronegative clusters improve stability and binding specificity of RNA-binding proteins . J Biol Chem 297 , 100945 ( 2021 ). doi: 10.1016/j.jbc.2021.100945 OpenUrl CrossRef PubMed ↵ Obayashi , T. , Kodate , S. , Hibara , H. , Kagaya , Y. & Kinoshita , K . COXPRESdb v8: an animal gene coexpression database navigating from a global view to detailed investigations . Nucleic acids research 51 , D80 ‒ D87 ( 2023 ). doi: 10.1093/nar/gkac983 OpenUrl CrossRef PubMed Yang , S. et al. COEXPEDIA: exploring biomedical hypotheses via co-expressions associated with medical subject headings (MeSH) . Nucleic acids research 45 , D389 ‒ D396 ( 2017 ). doi: 10.1093/nar/gkw868 OpenUrl CrossRef PubMed Zhu , Q. et al. Targeted exploration and analysis of large cross-platform human transcriptomic compendia . Nature methods 12 , 211 ‒ 214 , 213 p following 214 ( 2015 ). doi: 10.1038/nmeth.3249 OpenUrl CrossRef Miller , H. E. & Bishop , A. J. R . Correlation AnalyzeR: functional predictions from gene co-expression correlations . BMC Bioinformatics 22 , 206 ( 2021 ). doi: 10.1186/s12859-021-04130-7 OpenUrl CrossRef PubMed Raina , P. et al. GeneFriends: gene co-expression databases and tools for humans and model organisms . Nucleic acids research 51 , D145 ‒ D158 ( 2023 ). doi: 10.1093/nar/gkac1031 OpenUrl CrossRef PubMed Stuart , J. M. , Segal , E. , Koller , D. & Kim , S. K . A gene-coexpression network for global discovery of conserved genetic modules. Science (New York , N.Y .) 302 , 249 ‒ 255 ( 2003 ). doi: 10.1126/science.1087447 OpenUrl Abstract / FREE Full Text Lemoine , G. G. , Scott-Boyer , M. P. , Ambroise , B. , Perin , O. & Droit , A . GWENA: gene co-expression networks analysis and extended modules characterization in a single Bioconductor package . BMC Bioinformatics 22 , 267 ( 2021 ). doi: 10.1186/s12859-021-04179-4 OpenUrl CrossRef PubMed Warde-Farley , D. et al. The GeneMANIA prediction server: biological network integration for gene prioritization and predicting gene function . Nucleic acids research 38 , W214 ‒ 220 ( 2010 ). doi: 10.1093/nar/gkq537 OpenUrl CrossRef PubMed Web of Science ↵ Wong , A. K. , Krishnan , A. & Troyanskaya , O. G . GIANT 2.0: genome-scale integrated analysis of gene networks in tissues . Nucleic acids research 46 , W65 ‒ W70 ( 2018 ). doi: 10.1093/nar/gky408 OpenUrl CrossRef PubMed ↵ Tak Leung , R. W. , Jiang , X. , Chu , K. H. & Qin , J. ENPD - A Database of Eukaryotic Nucleic Acid Binding Proteins: Linking Gene Regulations to Proteins . Nucleic acids research 47 , D322 ‒ D329 ( 2019 ). doi: 10.1093/nar/gky1112 OpenUrl CrossRef PubMed ↵ Tasaki , S. , Gaiteri , C. , Mostafavi , S. & Wang , Y . Deep learning decodes the principles of differential gene expression . Nat Mach Intell 2 , 376 ‒ 386 ( 2020 ). doi: 10.1038/s42256-020-0201-6 OpenUrl CrossRef ↵ Shrikumar , A. , Greenside , P. & Kundaje , A. in Proceedings of the 34th International Conference on Machine Learning Vol. 70 (eds Precup Doina & Teh Yee Whye) 3145 ‒‒ 3153 (PMLR, Proceedings of Machine Learning Research, 2017 ). ↵ Ashburner , M. et al. Gene ontology: tool for the unification of biology. The Gene Ontology Consortium . Nature genetics 25 , 25 ‒ 29 ( 2000 ). doi: 10.1038/75556 OpenUrl CrossRef PubMed Web of Science ↵ Gene Ontology , C. , et al. The Gene Ontology knowledgebase in 2023 . Genetics 224 ( 2023 ). doi: 10.1093/genetics/iyad031 OpenUrl CrossRef PubMed ↵ Thomas , P. D. et al. PANTHER: Making genome-scale phylogenetics accessible to all . Protein Sci 31 , 8 ‒ 22 ( 2022 ). doi: 10.1002/pro.4218 OpenUrl CrossRef PubMed ↵ UniProt , C . UniProt: the Universal Protein Knowledgebase in 2025 . Nucleic acids research 53 , D609 ‒ D617 ( 2025 ). doi: 10.1093/nar/gkae1010 OpenUrl CrossRef PubMed ↵ Hu , M. et al. Evaluation of large language models for discovery of gene set function . Nature methods 22 , 82 ‒ 91 ( 2025 ). doi: 10.1038/s41592-024-02525-x OpenUrl CrossRef PubMed ↵ Zahra , K. , Dey , T. , Ashish Mishra , S. P. & Pandey , U. Pyruvate Kinase M2 and Cancer: The Role of PKM2 in Promoting Tumorigenesis . Front Oncol 10 , 159 ( 2020 ). doi: 10.3389/fonc.2020.00159 OpenUrl CrossRef PubMed Zhang , Z. et al. PKM2, function and expression and regulation . Cell Biosci 9 , 52 ( 2019 ). doi: 10.1186/s13578-019-0317-8 OpenUrl CrossRef PubMed Alquraishi , M. et al. Pyruvate kinase M2: A simple molecule with complex functions . Free Radic Biol Med 143 , 176 ‒ 192 ( 2019 ). doi: 10.1016/j.freeradbiomed.2019.08.007 OpenUrl CrossRef PubMed Azoitei , N. et al. PKM2 promotes tumor angiogenesis by regulating HIF-1alpha through NF-kappaB activation . Mol Cancer 15 , 3 ( 2016 ). doi: 10.1186/s12943-015-0490-2 OpenUrl CrossRef PubMed ↵ He , P. et al. PKM2 is a key factor to regulate neurogenesis and cognition by controlling lactate homeostasis . Stem Cell Reports 20 , 102381 ( 2025 ). doi: 10.1016/j.stemcr.2024.11.011 OpenUrl CrossRef PubMed Bonar , N. A. & Petersen , C. P . Integrin suppresses neurogenesis and regulates brain tissue assembly in planarian regeneration . Development 144 , 784 ‒ 794 ( 2017 ). doi: 10.1242/dev.139964 OpenUrl Abstract / FREE Full Text Porcheri , C. , Suter , U. & Jessberger , S . Dissecting integrin-dependent regulation of neural stem cell proliferation in the adult brain . J Neurosci 34 , 5222 ‒ 5232 ( 2014 ). doi: 10.1523/JNEUROSCI.4928-13.2014 OpenUrl Abstract / FREE Full Text Zheng , X. et al. Metabolic reprogramming during neuronal differentiation from aerobic glycolysis to neuronal oxidative phosphorylation . eLife 5 ( 2016 ). doi: 10.7554/eLife.13374 OpenUrl CrossRef PubMed Agostini , M. et al. Metabolic reprogramming during neuronal differentiation . Cell Death Differ 23 , 1502 ‒ 1514 ( 2016 ). doi: 10.1038/cdd.2016.36 OpenUrl CrossRef PubMed ↵ Saez , I. et al. Neurons have an active glycogen metabolism that contributes to tolerance to hypoxia . J Cereb Blood Flow Metab 34 , 945 ‒ 955 ( 2014 ). doi: 10.1038/jcbfm.2014.33 OpenUrl CrossRef PubMed ↵ Matthews , H. K. , Bertoli , C. & de Bruin , R. A. M . Cell cycle control in cancer . Nat Rev Mol Cell Biol 23 , 74 ‒ 88 ( 2022 ). doi: 10.1038/s41580-021-00404-3 OpenUrl CrossRef PubMed Dominguez-Brauer , C. et al. Targeting Mitosis in Cancer: Emerging Strategies . Molecular cell 60 , 524 ‒ 536 ( 2015 ). doi: 10.1016/j.molcel.2015.11.006 OpenUrl CrossRef PubMed ↵ Levine , M. S. & Holland , A. J . The impact of mitotic errors on cell proliferation and tumorigenesis . Genes Dev 32 , 620 ‒ 638 ( 2018 ). doi: 10.1101/gad.314351.118 OpenUrl Abstract / FREE Full Text ↵ Huang , X. et al. D-Serine regulates proliferation and neuronal differentiation of neural stem cells from postnatal mouse forebrain . CNS Neurosci Ther 18 , 4 ‒ 13 ( 2012 ). doi: 10.1111/j.1755-5949.2011.00276.x OpenUrl CrossRef PubMed ↵ El-Hattab , A. W. Serine biosynthesis and transport defects . Mol Genet Metab 118 , 153 ‒ 159 ( 2016 ). doi: 10.1016/j.ymgme.2016.04.010 OpenUrl CrossRef PubMed ↵ Kaida , A. & Iwakuma , T . Regulation of p53 and Cancer Signaling by Heat Shock Protein 40/J-Domain Protein Family Members . Int J Mol Sci 22 ( 2021 ). doi: 10.3390/ijms222413527 OpenUrl CrossRef ↵ Sterrenberg , J. N. , Blatch , G. L. & Edkins , A. L . Human DNAJ in cancer and stem cells . Cancer Lett 312 , 129 ‒ 142 ( 2011 ). doi: 10.1016/j.canlet.2011.08.019 OpenUrl CrossRef PubMed ↵ Yang , W. et al. ERK1/2-dependent phosphorylation and nuclear translocation of PKM2 promotes the Warburg effect . Nat Cell Biol 14 , 1295 ‒ 1304 ( 2012 ). doi: 10.1038/ncb2629 OpenUrl CrossRef PubMed Web of Science ↵ Israelsen , W. J. & Vander Heiden , M. G . Pyruvate kinase: Function, regulation and role in cancer . Semin Cell Dev Biol 43 , 43 ‒ 51 ( 2015 ). doi: 10.1016/j.semcdb.2015.08.004 OpenUrl CrossRef PubMed ↵ Osato , N. & Hamada , M . Systematic discovery of directional regulatory motifs associated with human insulator sites . bioRxiv ( 2025 ). ↵ Barrett , T. et al. NCBI GEO: archive for functional genomics data sets--update . Nucleic acids research 41 , D991 ‒ 995 ( 2013 ). doi: 10.1093/nar/gks1193 OpenUrl CrossRef PubMed Web of Science ↵ Edgar , R. , Domrachev , M. & Lash , A. E . Gene Expression Omnibus: NCBI gene expression and hybridization array data repository . Nucleic acids research 30 , 207 ‒ 210 ( 2002 ). doi: 10.1093/nar/30.1.207 OpenUrl CrossRef PubMed Web of Science ↵ Kolmykov , S. et al. GTRD: an integrated view of transcription regulation . Nucleic acids research 49 , D104 ‒ D111 ( 2021 ). doi: 10.1093/nar/gkaa1057 OpenUrl CrossRef PubMed ↵ Hinrichs , A. S. et al. The UCSC Genome Browser Database: update 2006 . Nucleic acids research 34 , D590 ‒ 598 ( 2006 ). doi: 10.1093/nar/gkj144 OpenUrl CrossRef PubMed Web of Science ↵ Mudge , J. M. et al. GENCODE 2025: reference gene annotation for human and mouse . Nucleic acids research 53 , D966 ‒ D975 ( 2025 ). doi: 10.1093/nar/gkae1078 OpenUrl CrossRef PubMed ↵ Van Nostrand , E. L. et al. A large-scale binding and functional map of human RNA-binding proteins . Nature 583 , 711 ‒ 719 ( 2020 ). doi: 10.1038/s41586-020-2077-3 OpenUrl CrossRef PubMed ↵ Hanzelmann , S. , Castelo , R. & Guinney , J . GSVA: gene set variation analysis for microarray and RNA-seq data . BMC Bioinformatics 14 , 7 ( 2013 ). doi: 10.1186/1471-2105-14-7 OpenUrl CrossRef PubMed ↵ Aibar , S. et al. SCENIC: single-cell regulatory network inference and clustering . Nature methods 14 , 1083 ‒ 1086 ( 2017 ). doi: 10.1038/nmeth.4463 OpenUrl CrossRef PubMed ↵ Avsec , Z. et al. Effective gene expression prediction from sequence by integrating long-range interactions . Nature methods 18 , 1196 ‒ 1203 ( 2021 ). doi: 10.1038/s41592-021-01252-x OpenUrl CrossRef PubMed ↵ Gasperini , M. et al. A Genome-wide Framework for Mapping Gene Regulation via Cellular Genetic Screens . Cell 176 , 1516 ( 2019 ). doi: 10.1016/j.cell.2019.02.027 OpenUrl CrossRef PubMed ↵ Kejiou , N. S. et al. Pyruvate Kinase M (PKM) binds ribosomes in a poly-ADP ribosylation dependent manner to induce translational stalling . Nucleic acids research 51 , 6461 ‒ 6478 ( 2023 ). doi: 10.1093/nar/gkad440 OpenUrl CrossRef PubMed ↵ Castello , A. et al. Insights into RNA biology from an atlas of mammalian mRNA-binding proteins . Cell 149 , 1393 ‒ 1406 ( 2012 ). doi: 10.1016/j.cell.2012.04.031 OpenUrl CrossRef PubMed Web of Science ↵ Toller-Kawahisa , J. E. et al. Metabolic reprogramming of macrophages by PKM2 promotes IL-10 production via adenosine . Cell Rep 44 , 115172 ( 2025 ). doi: 10.1016/j.celrep.2024.115172 OpenUrl CrossRef Yang , P. et al. Pyruvate kinase M2 accelerates pro-inflammatory cytokine secretion and cell proliferation induced by lipopolysaccharide in colorectal cancer . Cell Signal 27 , 1525 ‒ 1532 ( 2015 ). doi: 10.1016/j.cellsig.2015.02.032 OpenUrl CrossRef PubMed ↵ Liu , Z. , Le , Y. , Chen , H. , Zhu , J. & Lu , D . Role of PKM2-Mediated Immunometabolic Reprogramming on Development of Cytokine Storm . Front Immunol 12 , 748573 ( 2021 ). doi: 10.3389/fimmu.2021.748573 OpenUrl CrossRef PubMed ↵ Kachel , P. et al. Phosphorylation of pyruvate kinase M2 and lactate dehydrogenase A by fibroblast growth factor receptor 1 in benign and malignant thyroid tissue . BMC Cancer 15 , 140 ( 2015 ). doi: 10.1186/s12885-015-1135-y OpenUrl CrossRef ↵ Wang , J. et al. Down-regulation of NOTCH1 and PKM2 can inhibit the growth and metastasis of colorectal cancer cells . Am J Transl Res 14 , 5455 ‒ 5465 ( 2022 ). OpenUrl PubMed ↵ Dong , T. et al. Activation of GPR3-beta-arrestin2-PKM2 pathway in Kupffer cells stimulates glycolysis and inhibits obesity and liver pathogenesis . Nat Commun 15 , 807 ( 2024 ). doi: 10.1038/s41467-024-45167-5 OpenUrl CrossRef PubMed ↵ Gao , J. , Zhao , Y. , Li , T. , Gan , X. & Yu , H . The Role of PKM2 in the Regulation of Mitochondrial Function: Focus on Mitochondrial Metabolism, Oxidative Stress , Dynamic, and Apoptosis. PKM2 in Mitochondrial Function. Oxid Med Cell Longev 2022 , 7702681 ( 2022 ). doi: 10.1155/2022/7702681 OpenUrl CrossRef ↵ Liu , Z. , Le , Y. & Lu , D . PKM2: A novel helmsman of lipid metabolism . Cell Signal 134 , 111967 ( 2025 ). doi: 10.1016/j.cellsig.2025.111967 OpenUrl CrossRef ↵ Lee , K. C. Y. et al. PKM2 is a key regulator of cardiac lipid metabolism in mice . Mitochondrion 85 , 102070 ( 2025 ). doi: 10.1016/j.mito.2025.102070 OpenUrl CrossRef ↵ Li , L. , Zhang , Y. , Qiao , J. , Yang , J. J. & Liu , Z. R . Pyruvate kinase M2 in blood circulation facilitates tumor growth by promoting angiogenesis . J Biol Chem 289 , 25812 ‒ 25821 ( 2014 ). doi: 10.1074/jbc.M114.576934 OpenUrl Abstract / FREE Full Text ↵ Gomez-Escudero , J. et al. PKM2 regulates endothelial cell junction dynamics and angiogenesis via ATP production . Sci Rep 9 , 15022 ( 2019 ). doi: 10.1038/s41598-019-50866-x OpenUrl CrossRef PubMed ↵ He , D. et al. Methionine oxidation activates pyruvate kinase M2 to promote pancreatic cancer metastasis . Molecular cell 82 , 3045 ‒ 3060 e3011 ( 2022 ). doi: 10.1016/j.molcel.2022.06.005 OpenUrl CrossRef PubMed ↵ Zhang , X. , Lei , Y. , Zhou , H. , Liu , H. & Xu , P . The Role of PKM2 in Multiple Signaling Pathways Related to Neurological Diseases . Mol Neurobiol 61 , 5002 ‒ 5026 ( 2024 ). doi: 10.1007/s12035-023-03901-y OpenUrl CrossRef PubMed Venkatesh , H. S. et al. Electrical and synaptic integration of glioma into neural circuits . Nature 573 , 539 ‒ 545 ( 2019 ). doi: 10.1038/s41586-019-1563-y OpenUrl CrossRef PubMed ↵ Yu , L. , Chen , X. , Sun , X. , Wang , L. & Chen , S . The Glycolytic Switch in Tumors: How Many Players Are Involved? J Cancer 8 , 3430 ‒ 3440 ( 2017 ). doi: 10.7150/jca.21125 OpenUrl CrossRef PubMed ↵ Gallardo , M. et al. hnRNP K Is a Haploinsufficient Tumor Suppressor that Regulates Proliferation and Differentiation Programs in Hematologic Malignancies . Cancer Cell 28 , 486 ‒ 499 ( 2015 ). doi: 10.1016/j.ccell.2015.09.001 OpenUrl CrossRef PubMed ↵ Li , J. et al. HNRNPK maintains epidermal progenitor function through transcription of proliferation genes and degrading differentiation promoting mRNAs . Nat Commun 10 , 4198 ( 2019 ). doi: 10.1038/s41467-019-12238-x OpenUrl CrossRef PubMed ↵ Adelman , K. & Lis , J. T . Promoter-proximal pausing of RNA polymerase II: emerging roles in metazoans . Nature reviews. Genetics 13 , 720 ‒ 731 ( 2012 ). doi: 10.1038/nrg3293 OpenUrl CrossRef PubMed ↵ Abuhashem , A. , Garg , V. & Hadjantonakis , A. K . RNA polymerase II pausing in development: orchestrating transcription . Open Biol 12 , 210220 ( 2022 ). doi: 10.1098/rsob.210220 OpenUrl CrossRef PubMed ↵ Dang , H. et al. Oncogenic Activation of the RNA Binding Protein NELFE and MYC Signaling in Hepatocellular Carcinoma . Cancer Cell 32 , 101 ‒ 114 e108 ( 2017 ). doi: 10.1016/j.ccell.2017.06.002 OpenUrl CrossRef PubMed ↵ Liu , X. et al. Dynamic Change of Transcription Pausing through Modulating NELF Protein Stability Regulates Granulocytic Differentiation . Blood Adv 1 , 1358 ‒ 1367 ( 2017 ). doi: 10.1182/bloodadvances.2017008383 OpenUrl Abstract / FREE Full Text ↵ Robinson , D. C. L. et al. Negative elongation factor regulates muscle progenitor expansion for efficient myofiber repair and stem cell pool repopulation . Dev Cell 56 , 1014 ‒ 1029 e1017 ( 2021 ). doi: 10.1016/j.devcel.2021.02.025 OpenUrl CrossRef PubMed View the discussion thread. Back to top Previous Next Posted April 16, 2026. Download PDF Supplementary Material Data/Code Email Thank you for your interest in spreading the word about bioRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following Pathway redistribution reveals a shared signaling backbone and context-dependent regulatory modules in RNA-binding protein networks Message Subject (Your Name) has forwarded a page to you from bioRxiv Message Body (Your Name) thought you would like to see this page from the bioRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share Pathway redistribution reveals a shared signaling backbone and context-dependent regulatory modules in RNA-binding protein networks Naoki Osato , Kengo Sato bioRxiv 2025.03.03.641203; doi: https://doi.org/10.1101/2025.03.03.641203 Share This Article: Copy Citation Tools Pathway redistribution reveals a shared signaling backbone and context-dependent regulatory modules in RNA-binding protein networks Naoki Osato , Kengo Sato bioRxiv 2025.03.03.641203; doi: https://doi.org/10.1101/2025.03.03.641203 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Bioinformatics Subject Areas All Articles Animal Behavior and Cognition (7635) Biochemistry (17691) Bioengineering (13892) Bioinformatics (41937) Biophysics (21452) Cancer Biology (18589) Cell Biology (25504) Clinical Trials (138) Developmental Biology (13378) Ecology (19899) Epidemiology (2067) Evolutionary Biology (24320) Genetics (15609) Genomics (22506) Immunology (17736) Microbiology (40394) Molecular Biology (17181) Neuroscience (88605) Paleontology (666) Pathology (2832) Pharmacology and Toxicology (4824) Physiology (7641) Plant Biology (15156) Scientific Communication and Education (2045) Synthetic Biology (4294) Systems Biology (9825) Zoology (2271)

Text is read by the "Ask this paper" AI Q&A widget below. Extraction quality varies by source — PMC NXML preserves structure cleanly, OA-HTML may include some navigation residue, and OA-PDF can have broken hyphenation. The publisher copy (via DOI) is the canonical version.

My notes (saved in your browser only)

Ask this paper AI returns verbatim quotes from the full text · source: preprint-html

Answers must be backed by verbatim quotes from this paper's full text. Hallucinated quotes are dropped automatically; if no verbatim passage answers the question, we say so. How this works

Citation neighborhood (no data yet)

We don't have any in-corpus citations linked to this paper yet. This is a recent paper (2025) — citers typically take a year or two to land, and the OpenAlex reference graph may still be filling in.

Source provenance

europepmc
last seen: 2026-05-20T01:45:00.602351+00:00