Full text
75,104 characters
· extracted from
preprint-html
· click to expand
Coarse-grained Modeling of Stress Granule Structure and Dissolution with Small-molecule Compounds | bioRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-M677548'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search New Results Coarse-grained Modeling of Stress Granule Structure and Dissolution with Small-molecule Compounds View ORCID Profile Jay L. Kaplan , View ORCID Profile Michael A. Webb doi: https://doi.org/10.1101/2025.03.10.642463 Jay L. Kaplan 1 Department of Chemical and Biological Engineering, Princeton University Princeton , NJ 08544, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Jay L. Kaplan Michael A. Webb 1 Department of Chemical and Biological Engineering, Princeton University Princeton , NJ 08544, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Michael A. Webb For correspondence: mawebb{at}princeton.edu Abstract Full Text Info/History Metrics Supplementary material Preview PDF Abstract Stress granules are biomolecular condensates composed of RNA and proteins that form in response to stress; their dysregulation is implicated in neurode-generative diseases. In this study, we develop a minimal stress-granule model, comprised of RNA and six key proteins associated with neurodegenerative conditions, and study its characteristics using coarse-grained molecular dynamics simulations. We find that RNA is essential to form stable condensates in these biopolymer mixtures, while underlying protein-protein interactions result in heterogeneous, multiphasic architectures. Inspired by therapeutic applications, we then challenge the stability of these condensates in the presence of twenty distinct small molecules. Simulation-derived properties are used to effectively classify compounds as “dissolving” or “non-dissolving,” in strong agreement with experimental findings. Further analysis suggests that dissolving compounds disrupt stress granule structure by intercalating with RNA and inducing intra-condensate mixing. These insights advance understanding of stress granule stability and demonstrate modeling strategies for screening of therapeutic candidates. Introduction Stress Granules (SGs) are biomolecular condensates, membrane-less organelles that form through liquid-liquid phase separation, featuring ribonucleoprotein assemblies characterised by a vast proteomic and transcriptomic network ( 1 , 2 ) . SGs form in the cytoplasm upon exposure to environmental stressors and possess biphasic morphologies with dense cores surrounded by a more diffuse shell ( 3 ) . SG formation is thought to be driven by certain protein species that interact with mRNA and sequester it inside a condensate, thereby controlling mRNA metabolism and translation during cellular stress ( 4 , 5 ) by either storing mRNA, transferring it for degradation, or releasing it for polysomes to re-initiate translation ( 6 ) . Chronic cellular stress, mutations in key proteins, alterations in various mRNA transcripts, and sufficiently high concentrations of specific proteins, prevent dissolution of SGs and can cause the formation of aberrant, persistent SGs, which promote further aggregation of disease-related proteins. These aberrant SGs and the accompanying aggregation of disease-related proteins and mRNA transcripts have been linked to various diseases, including neurodegenerative disorders, cancers, and viral infections, making these aberrant condensates a promising target for novel therapeutics ( 7 ) . There is substantial interest in small-molecule compounds as prospective therapeutics that function by modulating the phase behavior and stability of biomolecular condensates, including SGs ( 8 ) . SG formation is thought to be driven by the macromolecular crowding of accumulated mRNA and proteins. This enables multivalent interactions between the mRNA and key client and scaffold proteins ( 9 ) . The excessive aggregation of specific proteins within SGs, such as TDP43 and FUS, can cause the transition of the condensate from a liquid-like to solid-like state. This inhibits disassembly and leads to subsequent chronic SG formation with consequent further aggregation of these disease-inducing biopolymers in the affected cells ( 10 ) . Therefore, understanding SG structure, dynamics, and stability could reveal many therapeutic targets ( 11 ) . A limited number of experimental studies have screened existing small-molecule libraries to determine which compounds might partition into, and modulate, SG stability ( 12 , 13 ) . Empirical observations suggest that particular physiochemical factors and mechanisms of interaction may be leveraged to promote SG dissolution. For example, many small-molecule compounds with notable “dissolving” effects on SGs empirically possess aromatic side chains ( 13 ) . Additionally, other small-molecule compounds apparently prevent TDP43 localization and accumulation into SGs as well as the formation of cytoplasmic puncta ( 13 ) , a hallmark of amyotrophic lateral sclerosis (ALS) ( 14 ) . Although there may be well-reasoned hypotheses regarding how and why certain small-molecule compounds effectively interact with SGs, direct evidence via experimental characterization can be challenging to obtain in such complex systems ( 15 ) . Developing principled and alternative methods to rapidly investigate the interactions of small-molecule compounds with SGs would bolster efforts in condensate-modulating therapeutic development. Molecular dynamics (MD) simulations can offer molecular-level insights into complex systems and phenomena, including phase-separating disordered proteins ( 16 , 17 ) . Due to the range of spatiotemporal scales in the underlying biophysics, multiscale simulation and coarse-grained (CG) modeling ( 18 , 19 ) have been particularly useful for providing insights into both single-chain and condensed-phase systems composed of intrinsically disordered proteins and RNA ( 20 – 30 ) . However, for residue-level or higher-resolution models, simulation of multicomponent biomolecular condensates, including stress-granule-like systems, is limited but could offer invaluable insights into their properties ( 31 ) . By extension, simulation studies on the interactions and effects of small-molecule compounds on biomolecular condensate properties are nascent. Therefore, the efficacy of using CG modeling to study SG structure and dissolution remains unexplored but offers significant applications in structural biology and high-throughput in silico screens for biomolecular condensate modulating therapeutics ( 8 ) . Here, we study the structure and stability of SGs, including their interactions with small-molecule compounds, using highly efficient data-driven CG modeling. We first generate a computationally tractable model with biophysical relevance, and similar composition to, stress granules. We examine the structural properties of the resulting simulated condensate. Using a simple and efficient data-driven coarse-graining parameterisation procedure, we then investigate how properties of SGs are impacted by a chemically diverse set of small-molecule compounds (SMCs). Our simulation results strongly correlate with experimental outcomes regarding the propensity of SMCs to modulate SG stability. Despite the small system size relative to in vivo SGs and the coarse-grained resolution afforded by the MPiPi model used in these simulations, we elucidate additional aspects of SG structure and identify possible novel mechanisms of SMC-induced stress-granule dissolution. This includes the intercalation of SMCs with RNA that decimates protein-RNA contacts which induces intra-condensate mixing and a collapse of the SG’s multiphasic architecture. Overall, this work provides new molecular-level understanding of stress-granule properties as well as a platform for rapidly screening and developing biomolecular condensate modulating de novo SMC therapeutics. Results Minimal mixtures of SG proteins and RNA yield stable condensates Stress granules are highly complex non-membrane-bound organelles with a diverse proteome consisting of hundreds of associated proteins ( 3 ) . Therefore, we first aim to identify a computationally tractable model for the core of a SG, which is comprised of both proteins and mRNA. For the proteins, we consider a reduced proteome comprised of only Tier 1 SG proteins, defined as protein species that possess high affinity to localize into SGs ( 32 ) . From this subset, we select proteins strongly implicated in neurodegenerative disorders ( 33 ) and then isolate only those proteins that have 800 or fewer amino acids (to reduce computational complexity). This results in a set of six proteins: G3BP1, FUS, TIA1, TTP, TDP43, and PABP1. Figure 1A shows the distribution of amino acids in these proteins, indicating a diverse range of possible interactions. Based on the relatively higher concentration of G3BP1 found in SG cores and its key role in SG formation ( 9 , 34 , 35 ) , we set the number of G3BP1 proteins to be approximately twice that of all other proteins. Download figure Open in new tab Figure 1: A mixture of key proteins and RNA form stable condensates, providing a simple model for the core of a stress-granule. (A) Distributions of amino acids in proteins present in the model stress granule: TTP, G3BP1, TIA1, TDP43, PABP1, and FUS. The amino acids are color-coded based on their functional group classification. (B) Distribution of nucleic acids in the modified ZACN poly-(A) mRNA transcript present in the model stress granule. (C) Examination of the fraction of biopolymers within the largest cluster, ,identified in systems with different mass fractions of protein, w p . The color of the bars also indicates the mass fraction of the protein present in the system with darker bars representing a larger w p . (D) Comparison of the fraction of biopolymers within the largest cluster, ,for different minimal SG models of a single protein species for systems with the optimal mass fraction of (w p = 0.5) and without (w p = 1.0) RNA. For the mRNA, we form a single, relatively short RNA transcript. In particular, we begin with the ZACN transcript, which exhibits a high affinity to localize into SGs ( 2 ) . We then extract only a randomized portion of the coding region and exclude the UTR regions to reduce the transcript length from 2,651 to 700 nucleic acids. We then add a 140bp poly-(A) tail, as SGs stain positive for poly-(A) mRNA ( 2 ) . Figure 1B shows the distribution of nucleic acids for the model RNA sequence. The simplifications described may be reasonable because the diverse SG transcriptome does not require specific mRNA transcripts or characteristics to phase separate ( 36 ) . However, RNA strand length and sequence can still influence the phaseseparating behaviour of RNP granules ( 2 , 37 , 38 ) , which motivates future consideration of how RNA composition and length impact the observations from computational SG models. To determine whether mixtures of the aforementioned biopolymers can form stable condensates, we perform MD simulations using the MPiPi force field ( 24 ) . This force field utilises an implicit solvent and treats amino and nucleic acids at a one-bead-per-residue resolution; interactions are modeled via bonded and non-bonded terms, including explicit electrostatics with screening (see also “Details of MD Simulations”). It is important to note that this model does not accurately represent complex structured protein regions but effectively represents IDPs and LLPS while allowing for running simulations of sufficient scale to investigate SG properties. Effectively modeling both structured domains and intrinsically disordered regions (IDRs) is an existing challenge for coarse-grained protein forcefields ( 39 ) and given the strong presence of IDPs in SGs we elected a model effective for the latter. Therefore, the results focus on the general propensity to phase separate based on the proportion and sequence of amino acids and on proteins with long Intrinsically Disordered Regions (IDRs). We quantify stability of a biopolymer mixture by the fraction of biopolymers found in the largest connected aggregate in the system, ,and the total number of individual aggregates, k ; the expectation is that more stable SGs will possess larger and lesser k . Figure 1C shows that stable condensates form over a range of intermediate protein mass fractions, w p (defined as the mass of protein in the system relative to the total mass of biopolymer). In pure RNA systems, (w p = 0.0), phase-separation of the RNA is limited, with .The negligible is due to transient interactions amongst RNA strands; RNA can also phase-separate in the absence of proteins under certain conditions ( 40 ) . However, once the system contains a nominal protein fraction (e.g., w p = 0.1), rises to ca. 0.48, approximately three-fold over the pure RNA systems. This indicates greater propensity for a protein-RNA mixture to induce liquid-liquid phase separation and that RNA is a key scaffold of SGs. Nearly all biopolymers interact to form a single large cluster for protein mass fractions of 0.4 ≤ w p ≤ 0.7. These values align with prior work using phenomenological models ( 41 ) , though some experiments suggest greater mRNA composition ( 2 ) . The disparity is possibly related to the use of our comparatively short RNA transcript which reduces the number of available sites for RNA-protein interactions and motivates further investigation on the effect of RNA length and composition on SG stability. Nevertheless, we characterise the system with w p = 0.5 as a “control” SG model, as it represents the most stable condensate and therefore a challenging test for dissolution by SMCs. Heterotypic protein-RNA interactions underlie SG formation To better understand drivers of stability in the SG model, we contrast separate systems consisting of each individual composite protein species and a system consisting of the full SG mixture both in the presence of RNA, specifically the optimal protein mass fraction, w p = 0.5, and in the absence of RNA, w p = 1.0. Figure 1D shows that RNA is essential to form the model SG, with for all protein species systems containing the optimal RNA mass fraction. This compared to systems in the absence of RNA that exhibit for five of the six protein systems, for the SG mixture, ,for the pure FUS system. These results broadly agree with prior observations and expectations regarding the role of mRNA in SGs ( 5 , 38 , 42 , 43 ) and also essential components in the SG proteome ( 3 , 9 , 34 ) . In particular, five of the six protein species do not centralize into a significant single cluster in the absence of RNA (i.e., w p = 1.0); the lone exception is FUS, which is known to phase-separate without RNA. Without RNA, the propensity for protein localization aligns well with the overall abundance of aromatic residues, particularly Tyrosine, which perhaps highlights the relevance of π − π interactions deemed vital for LLPS ( 24 ) . With RNA (i.e., w p = 0.5), virtually all biopolymers coalesce into a single droplet. Comparing the full SG system to individual protein species systems, the protein mixture (labeled as ‘SG’ in Fig. 1D ) exhibits the second highest propensity for aggregation without RNA (after FUS). This suggests that heterotypic (mixed species) protein interactions are a stronger driver of condensate formation than homotypic (same species) interactions, but that protein-RNA interactions are specifically impactful. SGs exhibit a hierarchically heterogeneous morphology Having established a stable, minimalist model SG, we now examine its structural characteristics. Figure 2A qualitatively illustrates the heterogeneous distribution of composite biopolymers within a representative configuration of the model SG. The structure possesses a dense protein-RNA mixture at the center surrounded by a more diffuse, RNA-rich shell. Figure 2B offers quantitative confirmation, examining the distribution of biopolymers and overall size of the model SG via radial density profiles (RDPs). The dense protein-RNA core has an overall concentration of 460 mg/mL and a protein concentration of 340 mg/mL, consistent with prior protein condensate literature ( 44 – 46 ) . The core concentration remains relatively constant up to about 100 Å before decreasing, signifying the transition between concentrated and dilute phases. The ratio of protein to RNA concentrations inverts around 250 Å (zoomed inset in Figure 2B ), indicating the transition to the diffuse RNA-rich shell. Download figure Open in new tab Figure 2: The model stress granule possesses a heterogeneous structure. (A) Representative simulation snapshot of the model stress granule system following equilibration. Proteins are generally found within a dense center. (B) Radial mass-density profiles resolved by biopolymer class. (C) normalised radial mass-density profiles resolved by biopolymer species. Each concentration curve is normalised by the maximum concentration of that species, and profiles are vertically shifted by ‘1’ for visual clarity. The dashed line represents the radial distance at which the normalised concentration reaches 0.5 determined using a fifth degree univariate spline. (D) RNA radial mass-density profiles resolved by nucleic acid type. Figure 2C indicates that different biopolymers preferentially localize to specific SG regions. This is revealed by tracking visual landmarks, such as the distances from the center of the SG at maximum and half-maximum concentrations. For example, the TDP43 concentration rapidly decreases, indicating its sequestration at the SG center. The FUS concentration decreases more gradually but remains more centrally localized than other proteins. Meanwhile, G3BP1 maintains a relatively constant concentration up to 90 Å, then gradually decreases to half-maximum at 160 Å. This is consistent with observations of TDP43 de-mixing from G3BP1 in binary mixtures ( 47 ) and of multiphasic architectures in two protein and RNA systems ( 28 ) but has not been observed in a system of this complexity. The other proteins – TIA1, PABP1, and TTP – have distinct maximum concentrations at distances under 200 Å. TIA1 peaks at 90 Å, just outside the center, while PABP1 and TTP reach their maxima between 130 Å and 150 Å, closer to the diffuse RNA boundary. Up to its half-maximum concentration at 170 Å, RNA distribution closely tracks G3BP1 but decays thereafter. This altogether suggests that the model SG has a dense core of heterogeneously distributed proteins mixed with RNA, surrounded by a diffuse, RNA-rich shell, indicating a multiphasic structure. Figure 2D , which resolves the RDP of RNA by nucleic acid type, reveals that the model RNA sequence is also heterogeneously distributed. Adenine concentration increases from the SG center to a maximum of 120 mg/mL at 90 Å before sharply decreasing. In contrast, uracil, cytosine, and guanine concentrations are lower than adenine at the center but exceed it at 180 Å. The data indicate that RNA adopts a preferential orientation, with the poly-(A) tail embedded in the dense protein-RNA core and the coding regions extending into the diffuse shell. This suggests that most protein-RNA interactions involve the poly-(A) tail, while the coding regions act as surfactants. This is supported by the slight difference in interfacial tension between the pure protein condensate (8.65 ·10 −2 ±3.72· 10 −6 mN/m) and the full SG model (8.49· 10 −2 ±3.58 ·10 −6 mN/m). Previous studies have shown that RNA length influences phase behavior, with shorter transcripts acting as surfactants and longer ones as stabilizers ( 37 ) . These simulations suggest that a single RNA strand can perform both roles depending on the partitioning of RNA segments between the core and shell. Overall, Fig. 2 highlights multiple levels of structural organization within the model SG. Biopolymer interactions are varied and specific in the SG To understand the formation of the hierarchical SG structure, we characterised the frequency of biopolomyer interactions using contact maps, resolved by species ( Fig. 3A,B ) and by residue type ( Fig. 3C,D ). Both raw ( Fig. 3A,C ) and normalised ( Fig. 3B,D ) contact maps are analysed, with the normalised maps revealing interaction trends based on affinity rather than species abundance. Figure 3 overall illustrates that varied interaction patterns amongst biopolymers underlie the observed multiphasic structure of the SG. Download figure Open in new tab Figure 3: Interaction patterns between protein-RNA and amino/nucleic underlie the structure of the model stress granule. (A) Heat map of unnormalised contact probabilities resolved by different species of biopolymers. (B) Heat map of normalised contact probabilities resolved by different species of biopolymers. (C) Heat map of unnormalised contact probabilities resolved by amino acids and nucleic acids. (D) Heat map of normalised contact probabilities resolved by amino acids and nucleic acids. In both (B) and (D) , the number of contacts are normalised as described in “Methods.” Examining interactions resolved by species, Fig. 3A supports the observation that RNA serves as a scaffold in the SG model. It is essential for the formation of the condensate and exhibits high contact probabilities with most proteins, except for TDP43 and TTP. We hypothesize that disruption of these RNA-protein interactions functions to dissolve SGs. G3BP1, a core SG network protein ( 9 , 34 , 35 ) , exhibits strong interactions with all biopolymers, especially with itself and RNA. TDP43, which localizes to the SG center, shows low contact probabilities overall but interacts more with G3BP1, RNA, and FUS, while having minimal interaction with TTP, which resides in the SG periphery. The normalised contact probabilities in Fig. 3B highlight that biopolymers have a higher affinity to engage in homotypic interactions but still exhibit significant heterotypic contacts with RNA that are necessary for condensate formation, supporting prior observations of intra-condensate de-mixing ( 28 , 47 ) . FUS shows the strongest homotypic interactions and significant contacts with RNA, consistent with its phase-separation behavior ( 44 , 48 ) . TDP43 also exhibits strong self-interactions and localizes centrally in the SG ( 47 ) ; a contact map resolved by index (Supplementary Information, Fig. S8A) shows high contact frequencies are seen among its IDRs, RRM regions, and NTD homodomains, consistent with previous studies ( 49 ) . Proteins like TIA1 and G3BP1 interact significantly with FUS and RNA, with G3BP1 driving phase-separation via poly-(A) RNA ( 34 , 50 ) . The RNA itself shows homotypic interactions, likely due to poly-(A) tail embedding, while PABP1 and TTP exhibit strong contacts with RNA in line with their known binding behaviors ( 51 ) . This reflects an intriguing balance, as the presence of both RNA and protein is critical for SG formation ( Fig. 1C ) and that multiple protein species systems more readily form condensates than single protein species systems ( Fig. 1D ), yet once formed, intra-condensate demixing results in a multiphasic architecture ( Fig. 2C ) arising predominantly from homotypic interactions ( Figs. 3A,B ). The residue-resolved contact maps in Figs. 3C,D suggest that SG structure is driven by key amino acids—Arg, Gly, Pro, Ser, Gln, Tyr—and nucleic acid A. These residues promote liquid-liquid phase separation via hydrogen bonding, π -bonding, and electrostatic interactions, consistent with prior findings ( 52 – 54 ) . Proteins in the SG model, such as FUS, TIA1, PABP1, TDP43, TTP, and G3BP1, are rich in these residues ( 50 , 55 – 58 ) , reinforcing their roles in phase separation. Strong contacts of Arg and Gly with adenine, prominent due to the poly-(A) tail, align with known RNA-binding motifs ( 59 , 60 ) , and normalised maps highlight the dominance of RNA-RNA C, G and U interactions. Together, this supports both roles of RNA: the poly-(A) tail embedded in the core serving as a scaffold and the coding regions in the diffuse shell serving as a surfactant. This is further illustrated in contact maps tracking residue index and region rather than type (Supplementary Information, Fig. S8B). Trp homotypic interactions, driven by π - π bonding, remain significant despite the low abundance of Trp. These findings, combined with adenine core localization, reflect the expected biophysical behavior of SGs, confirming the relevance of the model to previous experimental insights. Machine learning facilitates coarse-graining of small-molecule compounds We now evaluate the impact of small-molecule compounds (SMCs) on SG structure and properties. Based on previous experimental assays ( 13 , 61 ) , we consider ten SMCs expected to dissolve SGs and another ten that do not induce dissolution. A major obstacle for studying this effect is the absence of compatible CG force-fields and complexities associated with such developments ( 18 , 19 , 62 ) . As a feasible alternative, we employ a data-driven, though fundamentally approximate, approach to create SMC CG models that align with the resolution and interaction style of the MPiPi force-field. Consequently, the results should be interpreted as suggestive of what could be observed for a set of chemically distinct molecules. To obtain CG structures of these SMCs, we use the spectral graph-based coarse-graining algorithm ( 63 ) . The spectral graph-based coarse-graining scheme produces a series of hierarchically linked molecular representations guided by spectral properties of the chemical adjacency matrix and yielding reasonable representations relative to ad hoc assignments. Figure 4A and 4B respectively show the resulting structures for “dissolving” small molecules (DSMs) and the “non-dissolving” small molecules (NDSMs). Download figure Open in new tab Figure 4: Overview of models of small-molecule compounds. (A) Chemical structures and coarse-grained representations for the set of ten small-molecule compounds (DSM-1 - DSM-10) with expected propensity to dissolve stress granules in ascending order of molar mass. (B) Chemical structures and coarse-grained representations for the set of ten small-molecule compounds (NDSM-1 - NDSM-10) without expected propensity to dissolve stress granules in ascending order of molar mass. (C) Performance of random forest regression as a function of feature-selection number and force field parameter. Separate models are trained to predict the four different Wang-Frenkel CG interaction parameters. (D) Homotypic interaction curves predicted by random forest models for the small-molecule compounds split between the ten DSMs (top) and te NDSMs (bottom). To model interactions with the SMCs, a two-step parameterisation procedure is employed. The first step uses a random forest regression model to predict homotypic interaction parameters from a feature vector of chemical descriptors calculated using Mordred ( 64 ) . The model is trained on data from MPiPi, leveraging its pre-existing parameterisation of amino and nucleic acids, and exploits the chemical similarity between those residues and SMC structures through an optimised descriptor-to-parameter mapping. Figure 4C illustrates the capacity of the model to predict homotypic force-field parameters; models are trained on approximately 80% of the parameterised amino and nucleic acids (i.e., 19/24) using various feature sets and then tested on the remaining 20% (i.e., 5/24). All models perform well, with the R 2 values for every parameter exceeding 0.85 with a sufficiently large feature set. Additionally, in Figure 4D we see that the DSMs tend to have less variable interaction energies than NDSMs. The second step identifies an appropriate mixing rule for handling heterotypic interactions. The use of interaction mixing rules for determining the heterotypic interactions is sensible given this difference in homotypic interaction parameters. In the MPiPi force-field, all interactions, both homotypic and heterotypic, are explicitly parameterised. To prevent extrapolation and over-reliance on the model, we avoid explicit parameterisation for the SMCs and rather examined common parameter-mixing rules to determine how well they could reproduce the amino/nucleic acid heterotypic interaction parameters for MPiPi. Although no mixing rule performs universally well, the Lorentz-Berthelot mixing rules ( 65 , 66 ) most closely resemble the parameterised heterotypic interactions of the MPiPi force-field and were thus used for the analysis. Simulations effectively classify dissolving and non-dissolving small molecules We run MD simulations to measure the effect of each CG small-molecule on SG stability and structure. Figure 5A, B , and C illustrate that DSMs and NDSMs have distinct effects on various SG properties. Specifically, DSMs reduce the fraction of biopolymers in the largest cluster compared to both NDSMs and the SG control, suggesting that DSMs strip biopolymers from the cluster and induce SG dissolution. This is also clear when examining the proportion of each biopolymer species that remain part of the SG (Supplementary Information, Fig. S1). As a result, the number of contacts between biopolymers is higher for NDSMs than for DSMs (Supplementary Information, Fig. S2), owing to the fact that biopolymers are no longer as localized in the DSM-containing systems. Numerous other properties also display distinct responses between DSMs and NDSMs (see also Fig. S3-S4 in the Supplementary Information). Perhaps counterintuitively, the partition coefficient of the DSMs is lower than that of the NDSMs, which we hypothesise is due to the removal of biopolymers that interact most strongly with DSMs. Additionally, a marked increase in the diffusion coefficient of G3BP1 proteins in the largest cluster indicates that DSMs reduce macromolecular crowding and the viscosity of the SG. We note that the diffusion coefficients are substantially larger than those experimentally measured for biomolecular condensates ( 67 , 68 ) ; this is generally expected from CG models that do not specifically target reproduction of dynamical quantities, while trends may be preserved ( 18 , 24 , 69 ) . Download figure Open in new tab Figure 5: Simulations reveal distinct effects of dissolving and non-dissolving small-molecule compounds on SG properties. (A-C) Violin plot comparisons of the average behavior of systems containing dissolving and non-dissolving small-molecule compounds for (A) the fraction of biopolymers in the largest cluster , (B) the partition coefficient of the small-molecule into the SG P sm , and (C) the diffusion coefficient of G3BP1 in the largest cluster. The horizontal gray line is a reference to the average properties of the SG control system. The violin plots use kernel density estimation for the distributions but do not extend past the extreme data points. (D) Effective classification of small-molecule compound effects on stress-granules based on principle component analysis and k -means clustering. Each small-molecule compound is characterised by a 5-dimensional vector comprised of the analysed properties of the combined small-molecule/stress-granule system. The plot illustrates the two-dimensional projection resulting from PCA. Figure 5D demonstrates that examining how SMCs affect simulated SG properties can effectively classify DSMs and NDSMs. We perform standard scaling on five of the measured properties that best distinguished between the DSM and NDSM classes by removing the mean and scaling to unit variance. We then apply principal component analysis (PCA) to project these scaled features into a two-dimensional latent space for visualization and clustering analysis. This results in clear separation of DSMs (dark green) and NDSMs (light green), with the associated PCA eigenvalues (Supplementary Information, Fig. S5). Using then k -means clustering (k = 2) to assign labels achieves 85% classification accuracy, with a 20% false positive rate and 10% false negative rate. Several results are striking, given the approximate nature of the modeling. For example, most DSMs of lower molar mass, represented by a single CG bead, are classified correctly, as is NDSM-8, a chemical fragment of DSM-9. Some misclassifications, such as NDSM-10, are understandable, as NDSM-10 was initially identified as a DSM in an experimental screen but failed a counter-screen ( 61 ) , making it a challenging test case. Extracting properties from the MD simulations is also essential for accurate classification, as using only SMC homotypic interaction model parameters yielded a 55% classification accuracy, with an 80% false positive rate and 10% false negative rate. This highlights that the simulations, by capturing the physics of biomolecular condensates and interactions with SMCs, align well with experimental outcomes. Small-molecule compounds alter internal SG structure To further assess the differences in SG properties between DSMs and NDSMs, we compare their radial density profiles. Figures 6A and 6B show that DSMs reduce condensate radius relative to NDSMs (from ca. 138 Å for NDSMs to 116 Å for DSMs); NDSMs do not markedly affect the size of the condensate. Resolved by biopolymer species, Fig. 6C and D demonstrate that both DSMs and NDSMs affect biopolymer localization, with DSMs having a stronger effect. In the presence of DSMs ( Fig. 6C ), there are notable changes to biopolymer distributions. TDP43 mixes into outer layers and is depleted in the center, while FUS, TIA1, PABP1, and RNA shift inward. G3BP1 and TTP remain similar to the control, with more compact and mixed biopolymer profiles. In the presence of NDSMs ( Fig. 6D ), effects are directionally similar but less striking than DSMs, with results being closer to the control SG. This suggests NDSMs do not significantly alter the multiphasic structure of the SG but do induce TDP43 mixing into other layers. Relative to the control SG, the positions of the midpoint concentrations for the biopolymers are more tightly grouped with DSMs than for NDSMs, and the shifts are smaller for the latter. Overall, these results suggest that DSMs primarily interact at the SG boundary, interacting with RNA in the diffuse shell and consequently stripping away the RNA scaffold which collapses the multiphasic architecture leading to a denser, smaller SG that resists further partitioning. By contrast, NDSMs penetrate the core but do not disrupt overall biopolymer distribution. Download figure Open in new tab Figure 6: Simulations reveal distinct effects of dissolving and non-dissolving small molecules on biopolymer distribution in the SG. (A) Radial density profile of SG systems that contain dissolving small-molecule compounds resolved by biopolymer class. The RDP for the DSMs is also included. (B) Radial density profile of SG systems that contain non-dissolving small-molecule compounds resolved by biopolymer class. The RDP for the NDSMs is also included. The small-molecule concentration is plotted on the right y-axis. (C) normalised radial mass-density profiles resolved by species for SG systems that contain dissolving small-molecule compounds. (D) normalised radial mass-density profiles resolved by species for SG systems that contain non-dissolving small-molecule compounds. In both (C) and (D) , the RDPs for the control SG are shown as dotted lines for reference. Each concentration curve is normalised by the maximum concentration of that species, and profiles are vertically shifted by ‘1’ for visual clarity. The vertical dashed lines represent the radial distance at which the normalised concentration reaches 0.5 determined using a fifth degree univariate spline. Small-molecule compounds alter SG component interactions To elucidate the disparity in structural effects between DSMs and NDSMs, we contrast how biopolymers interact based on the difference between contact probability maps obtained for DSMs and for NDSMs, in the same fashion as Fig. 3 . At the biopolymer level, Fig. 8A shows that DSMs increase contacts with FUS and, to a lesser extent, PABP1, while reducing interactions involving G3BP1, TIA1, and TTP. Notably, DSMs also decrease RNA contact with G3BP1 and TIA1, both of which are central to the SG structure and rely on RNA for phase separation. This supports the hypothesis that DSMs disrupt the function of RNA as a scaffold within the SG network. TDP43 contact probabilities remain largely unchanged. However, when normalised by biopolymer abundance ( Fig. 8B ), DSMs seemingly enhance overall interaction affinity across all biopolymers. This likely reflects the fact that biopolymers remaining after DSM action are more tightly bound within the SG. The increase in normalised PABP1 contacts further suggests tighter binding, consistent with its shift toward the SG center under DSM influence ( Fig. 7C,D ). Download figure Open in new tab Figure 7: Dissolving and non-dissolving small molecules have distinct effects on biopolymer interactions. (A, B) Heat maps reflecting the difference in interactions between biopolymer species in systems with DSMs versus NDSMs using (A) unnormalised and (B) normalised contact probabilities. (C, D) Heat maps reflecting the difference in interactions between residues in systems with DSMs versus NDSMs using (C) unnormalised and (D) normalised contact probabilities. Normalisation is based on the number of biopolymer species or residues assigned to the stress-granule structure (see Methods). Download figure Open in new tab Figure 8: Dissolving small molecules (DSMs) compounds preferentially interact with RNA relative to non-dissolving small molecules (NDSMs). (A , B) Heat maps reflecting the difference in interactions between biopolymer species and DSMs versus NDSMs using (A) unnormalised and (B) normalised contact probabilities. (C , D) Heat maps reflecting the difference in interactions between residues and DSMs versus NDSMs using (C) unnormalised and (D) normalised contact probabilities. Normalisation is based on the number of biopolymer species or residues assigned to the stress-granule structure (see Methods). Residue-resolved interaction analysis ( Fig. 8C ) reveals a marked increase in Gly interactions with most amino acids and all nucleic acids. This is especially pronounced in FUS, which shows significantly elevated contact probabilities. After normalising for amino and nucleic acid content ( Fig. 8D ), we observe a general increase in interactions, supporting the hypothesis that the residual SG after dissolution consists of tightly bound biopolymers. Notably, several interactions associated with phase-separation, such as A-protein and RNA-RNA contacts, are enhanced. Altogether, these results suggest that DSMs selectively disrupt weaker interactions within the SG, leaving behind a more compact, strongly interacting group of biopolymers. Dissolving compounds preferentially interact with RNA Compared to NDSMs, DSMs display higher affinity for RNA and interact less with most other biopolymers ( Fig. 8A ). This aligns with earlier findings that DSMs primarily act at the SG periphery, stripping biopolymers but not penetrating the core, unlike NDSMs. When normalised by biopolymer number ( Fig. 8B ), DSMs exhibit stronger but more varied interactions with biopolymers compared to NDSMs, with the highest affinity for RNA followed by TDP43. The positive differences in Fig. 8B suggest that NDSMs behave more like inert penetrants than major disruptors of SG structure. Meanwhile, the strong DSM-RNA interactions support the hypothesis that DSMs disrupt the SG by intercalating with nucleic acids, weakening mRNA contacts that scaffold the SG. The interaction with TDP43 is notable, given its role in SG aging and pathologies, and is consistent with reports that aromatic compounds prevent TDP43 aggregation in SGs ( 13 ) . At the level of amino and nucleic acids ( Fig. 8C,D ), DSMs primarily interact with nucleic acids, especially RNA, and less frequently with Arg, Gly, Phe, Tyr, and Trp. The residues Gly and Arg are abundant in FUS, PABP1, and G3BP1, which are weaker targets of the modeled DSMs ( Fig. 8B ). Additionally, DSMs show lower affinity for ‘A’ compared to other nucleic acids, suggesting preferential interaction with the coding region at the SG periphery rather than internal poly-(A) tails. Overall, this indicates that DSMs act by targeting RNA and RNA-dependent proteins, which may subsequently promote SG dissolution. Discussion We characterised the structure and interactions of a computationally tractable, minimal stressgranule (SG) model, consisting of six protein species and RNA ( Fig. 1 ). Despite its simplicity, it yielded biophysics consistent with prior studies but also new insights. Analysis revealed hierarchical and heterogeneous structuring within the SG ( Fig. 2 ). Across the various biopolymer species, G3BP1 and RNA emerged as critical components ( Fig. 3 ). G3BP1 notably exhibited extensive interactions across structural regions of the SG, consistent with previous experimental findings ( 9 , 34 , 35 ) . Additionally, RNA was crucial for SG stability, with the new insight that it acts as both a scaffold and surfactant: its poly-(A) tail anchors in the SG core while the coding region forms a diffuse boundary, behaving like a surfactant ( Fig. 3 ). This suggests that SG formation is driven by protein-RNA interactions and that dissolution is caused by disrupting these contacts. These findings augment our understanding of how biopolymers likely distribute in SGs based on key interaction patterns. The uncovered mechanisms can aid in the targeted design of therapeutics that preferentially intercalate with RNA. We further demonstrated the effectiveness of a data-driven coarse-grained modeling approach to elucidate the condensate-modulating effects of small-molecule compounds. Using machine learning, we developed models for a diverse set of compounds based on 20 molecules with known impacts on stress-granule stability ( Fig. 4 ). This approach, which bypasses intensive parameterisation, enabled efficient simulations of the SG model in the presence of specific small molecules. Our simulations revealed clear distinctions between dissolving (DSMs) and non-dissolving small molecules (NDSMs) in their effects on SGs. Principal component analysis of key properties enabled effective classification of compounds with dissolving capabilities, remarkably achieving 85% classification accuracy when compared to experimental data ( Fig. 5 ). Interestingly, we found that NDSMs permeate the SG but do not significantly disrupt its structure, whereas DSMs disturb the hierarchical organization of the SG ( Fig. 6 ). Over time, DSMs strip biopolymers from the condensate, resulting in a smaller, more mixed structure ( Fig. 7 ), which we propose as a precursor to further dissolution. These effects are largely driven by strong interactions between DSMs, RNA, and TDP43 ( Fig. 8 ). These findings not only suggest potential mechanisms of action for DSMs but also present a promising methodological avenue for future therapeutic screening by coarse-grained simulation. While this work offers valuable insights into SG biophysics and small-molecule interactions, there are notable limitations and opportunities for future work. Despite the strong alignment with experimental outcomes, the modeling was approximate. In particular, the small-molecule coarse-graining procedure was designed to be illustrative of a efficient data-driven parameterization schemes of new compounds into existing forcefields, rather than fully validated. Future work may establish this approach in a more rigorous framework and potentially extend its applicability to other systems and applications. Moreover, while our model SG is reasonably complex relative to other computational works regarding biomolecular condensates, it still falls short of replicating the full complexity of real SGs. Bridging this gap, both through larger simulations that incorporate additional protein species and varying RNA transcripts as well as more chemical detail, will be an important pursuit for a holistic understanding of these organelles and their role in disease. Finally, our analysis and classification relies on proxy properties that correlate with experimental dissolution. Future studies may better link these short-term observations with actual SG disassembly and include a wider range of SMCs. Addressing timescale limitations will be key to refining understanding of SG dissolution mechanisms. Materials and Methods General MD simulation procedures All MD simulations were performed using the LAMMPS ( 70 ) simulation package and the MPiPi force-field ( 24 ) . MPiPi was selected for its complementary parameterization of RNA at the time of study. The MPiPi force-field uses an implicit solvent and represents each amino and nucleic acid as a single bead. CG particles interact via a combination of bonded and non-bonded energy terms. Electrostatic interactions were computed using a Coulomb term with Debye-Hückel screening for charged pairs within 35 Å, with a relative dielectric constant of 80 and a Debye screening length of 7.95 Å, corresponding to a monovalent salt concentration of 0.15 M. Short-range non-bonded interactions were modeled using a Wang–Frenkel potential ( 71 ) , with interactions beyond 25 Å neglected. The equations of motion were integrated using the velocity-Verlet algorithm with a 20 fs time step, and neighbor lists were updated every 10 steps. Unless otherwise specified, simulations were performed in the canonical ensemble at T = 300 K within a cubic simulation cell with fixed side lengths and periodic boundary conditions. Each dimension of the cell was set to 2.4 µm (approximately the length of an unfolded ZACN transcript) to minimize finite-size effects. Temperature was controlled using the Nosé-Hoover thermostat. System preparation The mRNA transcript and protein sequences were obtained from the UniProt database and converted into PDB files. Initial SG system preparation followed five steps. First, initial configurations for each biopolymer were generated by simulating isolated chains (without periodic boundary conditions) for 100 ns. The number of chains was selected to ensure the final system contained no more than 70,000 CG particles. Second, the final configurations from the previous step were randomly positioned and oriented within a cubic simulation cell. Third, the system was relaxed via a 2 ns microcanonical ensemble simulation, with a maximum displacement of 0.1 Å per particle per time step. Fourth, to promote condensate formation, biopolymer chains beyond 0.8 µm from the center of the simulation cell experienced a weak drag force of 0.2 kcal/(mol Å) for 2 ns; this procedure was used to mitigate long diffusion times for the chains to coalesce. We note that the concentration profiles in Fig. 2B-2D indicate a relatively compact structure for the SG, indicating that the SG formation arises more from subsequent simulation than this initial co-localization. Fifth, the system was then simulated for 160 ns, with the final 120 ns used to assess stability based on the fraction of biopolymers in the largest cluster ( Fig. 1C ). This procedure was equivalently applied for systems with protein mass fractions ranging from w p = 0.0 to w p = 1.0 in increments of 0.1, as well as for individual protein species alone (w p = 1.0) and for individual species with RNA at w p = 0.5. For simulations with small molecules, 8, 324 molecules were added to the previously prepared system containing all biopolymer species at w p = 0.5, yielding a concentration of 1 µM. The molecules were randomly inserted into the simulation cell in a region outside of the largest biopolymer cluster. In particular, the allowed insertion regions extended from the outside of a sphere with a radius of 0.6 µm centered in the simulation cell to the sides of the simulation cell. Systems were then simulated for 2 ns in the microcanonical ensemble, with the maximum displacement capped at 0.1 Å per particle per time step to relax configurations. This setup allows the small molecules to diffuse to the model SG for interaction as they would in vivo rather than spontaneously appearing within the condensate. Calculation of properties From the initial configurations, systems were simulated for 2 µs to calculate fifteen properties related to SG stability, structure, and dynamics. The specific properties, described in detail below and summarised in Table 1 , include radial density profiles, the number of biopolymer clusters in the system, the fraction of biopolymers present in the model SG, the radius of gyration and diameter of the model SG, partitioning coefficients, interfacial tension, diffusivity of G3BP1 within the SG, viscosity of the model SG, and contact probability maps. For all static and structural properties, the 2 µs trajectory was split into forty 50 ns segments. The first segment was discarded (treated as additional equilibration), and the remaining segments were used with bootstrap resampling to compute means and sample standard errors. The same starting configuration was used for all systems, but the long simulation times facilitate decorrelation and sampling configuration space. Errors for derived quantities were obtained using error propagation. For the model SG system alone, properties were mostly stable after an initial transient period (see Supplementary Material, Fig. S6-S7). In contrast, systems with small molecules, particularly DSMs, were more varied as the molecules diffused and interacted with SG components. Therefore, properties from these simulations do not represent equilibrium values, though that does not preclude use for distinguishing between different behaviors (i.e., dissolving versus non-dissolving). View this table: View inline View popup Download powerpoint Table 1: Summary and nomenclature of calculated properties The largest cluster of biopolymers in the system is associated with the model SG. To identify clusters, we employed ( 17 , 72 , 73 ) , where clusters are merged if any residue of a biopolymer in one cluster is within a distance of r c = 16 Å of any residue in another cluster. By tracking clusters throughout the trajectory, the number of unique clusters, k , the fraction of total biopolymers in the largest cluster, ,and the radius of gyration of the largest cluster, , were calculated. The SG structure was characterised using radial mass density profiles (RDPs) of biopolymers associated with the largest cluster ( 17 , 74 ) . These were represented through c(r) , the concentration (mg/mL) as function of r , the distance from the center-of-mass of the cluster. Data were fitted to a sigmoidal function where β, α, R (sg) , and W (sg) are fitting parameters. The spatial extent of the SG is associated with R (sg) , while W (sg) is its interfacial width. The concentrations in the dense (associated with the SG) and dilute phases were obtained by c (sg) = β + α , and c (dil) = β − α ; these concentration estimates agree with alternative estimates based on the mass of biopolymers assigned to the largest cluster and its approximate volume. A quantity related to percolation was computed via where is the radius of gyration for all components in the simulation cell. Using the concentration profiles, partitioning coefficients, for both biopolymers and small molecules, were computed by considering the ratio of component concentrations between dense and dilute phases using where i indicates the species (or set of species) included in the RDP. Quantities related to interfacial tension ( 17 , 75 ) were computed as and where δa = (a − R) and δb = (b − R) with a and b are axis lengths determined from principal component analysis of the mass distribution obtained as described in Ref. ( 17 ) . The mean of Eqs. ( 4 ) and ( 5 ) was used as a proxy for the interfacial tension, γ , for analysis. Diffusion coefficients of G3BP1 in the model SG are computed from mean-squared displacements (MSD) of G3BP1 assigned to the SG. In particular, samples of the MSD are generated over 200-ns intervals using bootstrap resampling with respect to both trajectory segments as well as biopolymer chain. The produced curves were scanned over segments of length δt = 10 ns to identify a diffusive regime; the 10-ns segment with the slope closest to unity on log(⟨MSD⟩) versus log (t) plot is indicated by δt fit . Using a model ⟨MSD (t) ⟩ = mt + b and performing least-squares regression on data in this regime yields an estimate for diffusivity as where ⟨ m(δt fit )⟩ is an average of slopes fit from different bootstrapped samples. Quantities related to the viscosity of the SG were computed using two different methods. The first used the Stokes-Einstein equation with D as calculated above and as the average radius of gyration of G3BP1 in the SG; errors were estimated using error propagation. The second used the Green-Kubo relationship where P αβ (t) is an off-diagonal element of the pressure tensor, including contributions from all particles assigned to the largest cluster. Errors were estimated using bootstrap resampling with trajectory segments of 20 ns for computing the autocorrelation function. It is worth noting that while the right-hand side of Eq. ( 8 ) is well-defined in the CG simulations, it does not directly correspond to the physical viscosity of a condensate due to the absence of explicit solvent degrees of freedom ( 76 ) . Both Eq. ( 7 ) and ( 8 ) would be expected to be lower than experimentally measured viscosities. The connectivity in the SG was analysed using both unnormalised and normalised biopolymer and amino/nucleic acid contact maps. For contacts between biopolymers, the average number of contacts for each pair of biopolymers, ,was determined. A contact was defined when at least two monomers from different chains were within a cutoff distance of r c ≤ 16 Å; this cutoff is approximately twice the average σ of the MPiPi force-field. Contacts between amino/nucleic acids were similarly defined, except contacts were not counted for any amino acids separated by fewer than three bonds or any nucleic acids separated by fewer than four bonds. Contact probability maps were then computed as and and the notation indicates a normalised quantity, accounting for the abundance of interacting components, which is denoted as n i . Interactions were also quantified with an intermolecular contact map ( 49 ) via where N ij is the number of contacts at each i, j residue position, averaged over all chains for each biopolymer species, and max N ij is the maximum observed value. Differences in contact probability maps were used to highlight interaction effects induced by small molecules. We specifically report and where the superscript indicators of ‘(+dsm)’ and ‘(+ndsm)’ reference conditions upon addition of DSMs or NDSMs. Small-molecule selection To examine whether our model SG was sensitive to small molecule interactions, a set of experimentally tested compounds were selected based on their propensity to dissolve SGs ( 13 , 61 , 77 ) . In particular, twenty small molecules were chosen, evenly split between dissolving (DSM) and non-dissolving (NDSM) classes. The ten DSMs were the only small molecules in experimental assays found to dissolve SGs without targeting active biological processes ( 13 ) ; these were labeled D1-D10. For comparison, the ten NDSMs (ND1-ND10) were selected with the following rationale. ND1 was included as it was was a control in experimental studies. The set ND2, ND3, ND4, ND6, ND7, ND8, and ND9 were selected because they are chemical fragments of D4, D9, D8, and D10 ( 61 ) , making them challenging test cases. ND5 (1,6-hexanediol) was included to assess whether our analyses and methods could differentiate compounds effective at low versus high concentrations, as it disrupts SG stability only at high concentrations. ND10 was included because it was identified as a hit in an initial screen but failed to inhibit SG formation in a dose-dependent manner during a counterscreen ( 13 ) . Small-molecule coarse-graining The Graph-Based Coarse-Graining (GBCG) algorithm with spectral grouping was used to generate CG representations of all small-molecule compounds ( 63 ) . This spectral grouping approach successively combines nodes in a molecular graph based on eigenvector centrality rankings. Four iterations of grouping were performed, using adjacency matrices that incorporated both connectivity and node mass. To ensure no CG bead exceeded the molecular weight of the largest amino acid (Trp), nodes were merged only if the resulting mass remained below this 205 Da. This procedure produces a set of topologically CG beads to represent each small-molecule at a resolution consistent with the rest of the MPiPi force field. A data-driven parameterisation approach was employed to obtain force-field parameters for the derived CG beads. Each CG bead was assigned a SMILES string representing its atomistic structure and connectivity ( 78 ) , which was used to generate a feature vector of Mordred descriptors ( 64 ) . The same procedure was applied to the twenty amino acids and four nucleic acids in the MPiPi force field ( 24 ) . A random forest regression model was trained using scikit-learn ( 79 ) to predict MPiPi homotypic parameters from these feature vectors. A greedy forward-selection method based on maximum feature variance was employed to determine a relevant subset of descriptors: each descriptor with the highest remaining variance was added to the feature set if it improved the model’s coefficient of determination (R 2 ) on a test set; otherwise, it was discarded. This process was repeated until all descriptors were considered. A train-test split, comprising 19 amino/nucleic acids for training and 5 for testing (approximately an 80/20 split), was used to assess model performance. The results are reported in Supplementary Information, Table S1. Finally, the trained model was used to predict homotypic parameters for each CG bead of the small molecules. The parameters obtained by this approach are reported in the Supplementary Information, Table S2. Parameters for bonded interactions are reported in Supplementary Information, Table S3. Heterotypic interaction parameters between small molecules and between small molecules and amino/nucleic acids were specified using mixing rules. Although the MPiPi force field employs explicit heterotypic parameters, this approach would be labor intensive for the small molecules. Thus, we sought to identify an appropriate mixing rule that could approximate heterotypic parameters from homotypic parameters. Four widely used mixing rules were examined: Lorentz-Berthelot ( 65 , 66 ) , Waldman-Hagler ( 80 ) , Fender-Halsey ( 81 ) , and Kong ( 82 ) rules. Each rule was evaluated according to the sum of squared residuals between the explicit MPiPi heterotypic parameters and those derived from the mixing rules. The Lorentz-Berthelot rules minimized this metric, resulting in σ ij and R ij determined as arithmetic means and ϵ ij , µ ij , and ν ij as geometric means of the homotypic parameters. Funding This work was partially supported by the National Science Foundation grant 2237470 (M.A.W) Author contributions Conceptualization: J.L.K., M.A.W. Methodology: J.L.K., M.A.W. Investigation: J.L.K. Visualization: J.L.K., M.A.W. Supervision: M.A.W. Writing (original draft): J.L.K., M.A.W. Writing (review and editing): J.L.K., M.A.W. Competing interests Authors declare that they have no competing interests. Data and materials availability All data needed to evaluate the conclusions in the paper are available in the main text or the supplementary materials. Additional (raw) data not distributed may be requested from the corresponding author. Acknowledgments The authors thank Dr. Clifford P. Brangwyne, Dr. Anita Ðonlić, Mr. Satyen Dhamankar, Dr. Jerelle Joseph, and Dr. Gerhard Hummer for helpful discussions. Authors also acknowledge assistance from Princeton Research Computing at Princeton University, which is a consortium led by the Princeton Institute for Computational Science and Engineering (PICSciE) and Office of Information Technology’s Research Computing. References 1. ↵ D. S. Protter , R. Parker , Trends in Cell Biology 26 , 668 – 679 ( 2016 ). OpenUrl CrossRef PubMed 2. ↵ A. Khong , et al. , Molecular Cell 68 , 808 ( 2017 ). OpenUrl CrossRef PubMed 3. ↵ S. Jain , et al. , Cell 164 , 487 – 498 ( 2016 ). OpenUrl CrossRef PubMed 4. ↵ S. Hofmann , N. Kedersha , P. Anderson , P. Ivanov , Biochimica et Biophysica Acta (BBA) - Molecular Cell Research 1868 , 118876 ( 2021 ). OpenUrl CrossRef PubMed 5. ↵ D. Campos-Melo , Z. C. Hawley , C. A. Droppelmann , M. J. Strong , Frontiers in Cell and Developmental Biology 9 , 621779 ( 2021 ). OpenUrl CrossRef 6. ↵ N. Kedersha , et al. , Journal of Cell Biology 151 , 1257 – 1268 ( 2000 ). OpenUrl Abstract / FREE Full Text 7. ↵ B. Wolozin , P. Ivanov , Nature Reviews Neuroscience 20 , 649 – 666 ( 2019 ). OpenUrl CrossRef PubMed 8. ↵ D. M. Mitrea , M. Mittasch , B. F. Gomes , I. A. Klein , M. A. Murcko , Nature Reviews Drug Discovery 21 , 841 – 862 ( 2022 ). OpenUrl CrossRef PubMed 9. ↵ D. W. Sanders , et al. , Cell 181 , 306 ( 2020 ). OpenUrl CrossRef PubMed 10. ↵ A. Patel , et al. , Cell 162 , 1066 – 1077 ( 2015 ). OpenUrl CrossRef PubMed 11. ↵ Y. Shin , C. P. Brangwynne , Science 357 , eaaf4382 ( 2017 ). OpenUrl Abstract / FREE Full Text 12. ↵ I. A. Klein , et al. , Science 368 , 1386 – 1392 ( 2020 ). OpenUrl Abstract / FREE Full Text 13. ↵ M. Y. Fang , et al. , Neuron 103 , 802 – 819 ( 2019 ). OpenUrl CrossRef PubMed 14. ↵ B. A. Keller , et al. , Acta Neuropathologica 124 , 733 ( 2012 ). OpenUrl CrossRef PubMed Web of Science 15. ↵ S. Alberti , A. Gladfelter , T. Mittag , Cell 176 , 419 ( 2019 ). OpenUrl CrossRef PubMed 16. ↵ L. M. Pietrek , L. S. Stelzl , G. Hummer , Journal of Chemical Theory and Computation 16 , 725 ( 2019 ). OpenUrl 17. ↵ Z. Benayad , S. von Bülow , L. S. Stelzl , G. Hummer , Journal of Chemical Theory and Computation 17 , 525 – 537 ( 2020 ). OpenUrl 18. ↵ S. Dhamankar , M. A. Webb , Journal of Polymer Science 59 , 2613 – 2643 ( 2021 ). OpenUrl CrossRef 19. ↵ S. Shi , L. Zhao , Z.-Y. Lu , The Journal of Physical Chemistry Letters pp. 7280 – 7287 ( 2024 ). 20. ↵ G. L. Dignon , W. Zheng , J. Mittal , Current Opinion in Chemical Engineering 23 , 92 – 98 ( 2019 ). OpenUrl CrossRef PubMed 21. G. L. Dignon , W. Zheng , Y. C. Kim , R. B. Best , J. Mittal , PLoS computational biology 14 , e1005941 ( 2018 ). OpenUrl CrossRef 22. G. L. Dignon , W. Zheng , R. B. Best , Y. C. Kim , J. Mittal , Proceedings of the National Academy of Sciences 115 , 9929 ( 2018 ). OpenUrl Abstract / FREE Full Text 23. G. Tesei , T. K. Schulze , R. Crehuet , K. Lindorff-Larsen , Proceedings of the National Academy of Sciences 118 , e2111696118 ( 2021 ). OpenUrl Abstract / FREE Full Text 24. ↵ J. A. Joseph , et al. , Nature Computational Science 1 , 732 – 743 ( 2021 ). OpenUrl CrossRef PubMed 25. P. Y. Chew , J. A. Joseph , R. Collepardo-Guevara , A. Reinhardt , Biophysical Journal 122 , 295a ( 2023 ). OpenUrl 26. W. Zheng , et al. , The Journal of Physical Chemistry B 124 , 11671 – 11679 ( 2020 ). OpenUrl CrossRef PubMed 27. D. Aierken , J. A. Joseph , bioRxiv p. 2023.12.23.573204 ( 2023 ). 28. ↵ P. Y. Chew , J. A. Joseph , R. Collepardo-Guevara , A. Reinhardt , Biophysical Journal p. 1 – 14 ( 2023 ). 29. G. Valdes-Garcia , L. Heo , L. J. Lapidus , M. Feig , Journal of Chemical Theory and Com-putation 19 , 669 – 678 ( 2023 ). OpenUrl CrossRef 30. ↵ Y. An , M. A. Webb , W. M. Jacobs , Science Advances 10 , eadj2448 ( 2024 ). OpenUrl CrossRef PubMed 31. ↵ J. Jung , C. Tan , Y. Sugita , Nature Communications 15 , 3370 ( 2024 ). OpenUrl CrossRef PubMed 32. ↵ S. R. Millar , et al. , Chemical Reviews p. 9036–9064 ( 2023 ). 33. ↵ M. R. Asadi , et al. , Frontiers in Aging Neuroscience 13 , 650740 ( 2021 ). OpenUrl CrossRef PubMed 34. ↵ P. Yang , et al. , Cell 181 , 325 ( 2020 ). OpenUrl CrossRef PubMed 35. ↵ J. Guillén-Boixet , et al. , Cell 181 , 346 ( 2020 ). OpenUrl CrossRef PubMed 36. ↵ R. R. Edupuganti , et al. , Nature Structural & Molecular Biology 24 , 870 – 878 ( 2017 ). OpenUrl CrossRef PubMed 37. ↵ I. Sanchez-Burgos , L. Herriott , R. Collepardo-Guevara , J. R. Espinosa , Biophysical Journal 122 , 2973 – 2987 ( 2023 ). OpenUrl CrossRef PubMed 38. ↵ I. Sanchez-Burgos , J. R. Espinosa , J. A. Joseph , R. Collepardo-Guevara , PLOS Computational Biology 18 , e1009810 ( 2022 ). OpenUrl CrossRef PubMed 39. ↵ A. P. Latham , B. Zhang , Current opinion in structural biology 72 , 63 ( 2022 ). OpenUrl CrossRef PubMed 40. ↵ D. Aierken , J. A. Joseph , Journal of Chemical Theory and Computation 20 , 10209 ( 2024 ). Publisher: American Chemical Society . OpenUrl CrossRef 41. ↵ J. A. Joseph , et al. , Biophysical Journal 120 , 1219 – 1230 ( 2021 ). OpenUrl CrossRef PubMed 42. ↵ B. Van Treeck , et al. , Proceedings of the National Academy of Sciences 115 , 2734 – 2739 ( 2018 ). OpenUrl Abstract / FREE Full Text 43. ↵ K. Begovich , J. E. Wilhelm , Molecular Cell 79 , 991 ( 2020 ). OpenUrl CrossRef PubMed 44. ↵ A. C. Murthy , et al. , Nature Structural & Molecular Biology 26 , 637 – 648 ( 2019 ). OpenUrl CrossRef PubMed 45. J. P. Brady , et al. , Proceedings of the National Academy of Sciences 114 , E8194 ( 2017 ). OpenUrl Abstract / FREE Full Text 46. ↵ V. H. Ryan , et al. , Molecular Cell 69 , 465 ( 2018 ). OpenUrl CrossRef PubMed 47. ↵ X. Yan , et al. , bioRxiv p. 2024.01.23.576837 ( 2024 ). 48. ↵ L. R. Ganser , et al. , Structure 32 , 177 ( 2024 ). OpenUrl CrossRef 49. ↵ P. Mohanty , A. Rizuan , Y. C. Kim , N. L. Fawzi , J. Mittal , Protein Science 33 , e4891 ( 2024 ). OpenUrl CrossRef PubMed 50. ↵ H. Sidibé , A. Dubinski , C. Vande Velde , Journal of Neurochemistry 157 , 944 – 962 ( 2021 ). OpenUrl CrossRef PubMed 51. ↵ N. L. Kedersha , M. Gupta , W. Li , I. Miller , P. Anderson , The Journal of Cell Biology 147 , 1431 – 1442 ( 1999 ). OpenUrl Abstract / FREE Full Text 52. ↵ A. C. Murthy , et al. , Nature Structural & Molecular Biology 28 , 923 – 935 ( 2021 ). OpenUrl CrossRef PubMed 53. J. Li , et al. , Molecular Biomedicine 3 , 13 ( 2022 ). OpenUrl CrossRef PubMed 54. ↵ B. Wang , et al. , Signal Transduction and Targeted Therapy 6 , 290 ( 2021 ). OpenUrl CrossRef 55. ↵ Z. Monahan , et al. , The EMBO Journal 36 , 2951 – 2967 ( 2017 ). OpenUrl Abstract / FREE Full Text 56. R. Sawazaki , et al. , Scientific Reports 8 , 1455 ( 2018 ). OpenUrl CrossRef PubMed 57. H.-M. Chien , C.-C. Lee , J. J.-T. Huang , International Journal of Molecular Sciences 22 , 8213 ( 2021 ). OpenUrl CrossRef PubMed 58. ↵ V. D. Maciej , et al. , Nucleic Acids Research 50 , 10665 – 10679 ( 2022 ). OpenUrl CrossRef PubMed 59. ↵ M. Kiledjian , G. Dreyfuss , The EMBO Journal 11 , 2655 – 2664 ( 1992 ). OpenUrl CrossRef PubMed Web of Science 60. ↵ P. A. Chong , R. M. Vernon , J. D. Forman-Kay , Journal of Molecular Biology 430 , 4650 – 4665 ( 2018 ). OpenUrl CrossRef PubMed 61. ↵ R. J. Wheeler , et al. , bioRxiv p. p. 2024.01.05.721001 ( 2024 ). 62. ↵ W. G. Noid , The Journal of Physical Chemistry B 127 , 4174 ( 2023 ). OpenUrl CrossRef PubMed 63. ↵ M. A. Webb , J.-Y. Delannoy , J. J. de Pablo , Journal of Chemical Theory and Computation 15 , 1199 – 1208 ( 2018 ). OpenUrl 64. ↵ H. Moriwaki , Y.-S. Tian , N. Kawashita , T. Takagi , Journal of Cheminformatics 10 , 4 ( 2018 ). OpenUrl CrossRef PubMed 65. ↵ H. A. Lorentz , Annalen der Physik 248 , 127 – 136 ( 1881 ). OpenUrl CrossRef 66. ↵ D. Berthelot , Comptes rendus hebdomadaires des séances de l’Académie des Sciences 126 , 1703 – 1855 ( 1898 ). OpenUrl 67. ↵ N. O. Taylor , M.-T. Wei , H. A. Stone , C. P. Brangwynne , Biophysical Journal 117 , 1285 – 1300 ( 2019 ). OpenUrl CrossRef PubMed 68. ↵ L. Hubatsch , et al. , eLife 10 , e68620 ( 2021 ). OpenUrl CrossRef PubMed 69. ↵ J. F. Rudzinski , Computation 7 , 42 ( 2019 ). OpenUrl CrossRef 70. ↵ A. P. Thompson , et al. , Comp. Phys. Comm . 271 , 108171 ( 2022 ). OpenUrl CrossRef 71. ↵ X. Wang , S. Ramírez-Hinestrosa , J. Dobnikar , D. Frenkel , Physical Chemistry Chemical Physics 22 , 10624 – 10633 ( 2020 ). OpenUrl CrossRef PubMed 72. ↵ N. Michaud-Agrawal , E. J. Denning , T. B. Woolf , O. Beckstein , Journal of Computational Chemistry 32 , 2319 – 2327 ( 2011 ). OpenUrl CrossRef PubMed 73. ↵ R. J. Gowers , et al. , Proceedings of the 15th Python in Science Conference pp. 98–105 ( 2019 ). 74. ↵ J. Mittal , G. Hummer , Proceedings of the National Academy of Sciences 105 , 20130 – 20135 ( 2008 ). OpenUrl Abstract / FREE Full Text 75. ↵ J. Henderson , J. Lekner , Molecular Physics 36 , 781 – 789 ( 1978 ). OpenUrl CrossRef Web of Science 76. ↵ H. Liang , M. A. Webb , M. Chawathe , D. Bendejacq , J. J. de Pablo , Macromolecules 56 , 177 ( 2022 ). OpenUrl 77. ↵ F. Wang , J. Li , S. Fan , Z. Jin , C. Huang , Pharmacological Research 161 , 105143 ( 2020 ). OpenUrl CrossRef PubMed 78. ↵ N. M. O’Boyle , et al. , Journal of Cheminformatics 3 , 33 ( 2011 ). OpenUrl CrossRef PubMed 79. ↵ F. Pedregosa , et al. , Journal of Machine Learning Research 12 , 2825 ( 2011 ). OpenUrl 80. ↵ M. Waldman , A. Hagler , Journal of Computational Chemistry 14 , 1077 – 1084 ( 1993 ). OpenUrl CrossRef 81. ↵ B. E. Fender , G. D. Halsey , The Journal of Chemical Physics 36 , 1881 – 1888 ( 1962 ). OpenUrl CrossRef Web of Science 82. ↵ C. L. Kong , The Journal of Chemical Physics 59 , 2464 – 2467 ( 1973 ). OpenUrl CrossRef Web of Science View the discussion thread. Back to top Previous Next Posted March 13, 2025. Download PDF Supplementary Material Email Thank you for your interest in spreading the word about bioRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following Coarse-grained Modeling of Stress Granule Structure and Dissolution with Small-molecule Compounds Message Subject (Your Name) has forwarded a page to you from bioRxiv Message Body (Your Name) thought you would like to see this page from the bioRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share Coarse-grained Modeling of Stress Granule Structure and Dissolution with Small-molecule Compounds Jay L. Kaplan , Michael A. Webb bioRxiv 2025.03.10.642463; doi: https://doi.org/10.1101/2025.03.10.642463 Share This Article: Copy Citation Tools Coarse-grained Modeling of Stress Granule Structure and Dissolution with Small-molecule Compounds Jay L. Kaplan , Michael A. Webb bioRxiv 2025.03.10.642463; doi: https://doi.org/10.1101/2025.03.10.642463 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Biophysics Subject Areas All Articles Animal Behavior and Cognition (7622) Biochemistry (17648) Bioengineering (13871) Bioinformatics (41880) Biophysics (21423) Cancer Biology (18561) Cell Biology (25461) Clinical Trials (138) Developmental Biology (13364) Ecology (19866) Epidemiology (2067) Evolutionary Biology (24290) Genetics (15590) Genomics (22475) Immunology (17713) Microbiology (40328) Molecular Biology (17148) Neuroscience (88473) Paleontology (666) Pathology (2827) Pharmacology and Toxicology (4816) Physiology (7635) Plant Biology (15114) Scientific Communication and Education (2044) Synthetic Biology (4286) Systems Biology (9815) Zoology (2268)
Text is read by the "Ask this paper" AI Q&A widget below.
Extraction quality varies by source — PMC NXML preserves structure
cleanly, OA-HTML may include some navigation residue, and OA-PDF can
have broken hyphenation. The publisher copy
(via DOI)
is the canonical version.