Undersampling techniques for non-linear chemical space visualization

preprint OA: closed
📄 Open PDF Full text JSON View at publisher

Abstract

The visualization of high-dimensional chemical space is a critical tool for understanding molecular diversity, structure–property relationships, and for guiding compound selection. However, the performance of non-linear dimensionality reduction (DR) techniques like t-Stochastic Neighborhood Embedding (t-SNE), Uniform Manifold Approximation and Projection (UMAP), and Generative Topographic Mapping (GTM) are often susceptible to the choice of hyperparameters, along with the high cost of their training for large datasets. In this study, we investigated the effect of undersampling methods on the choice of hyperparameter selection for these non-linear dimensionality reduction methods. Our results demonstrate that selecting small representative subsets of chemical data not only reduces computational costs associated with hyperparameter training but also serves as an innovative means to train non-linear DR methods, leading to projections that better preserve the local structure within the chemical space.
Full text 41,177 characters · extracted from preprint-html · click to expand
Undersampling techniques for non-linear chemical space visualization | bioRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-M677548'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search New Results Undersampling techniques for non-linear chemical space visualization View ORCID Profile Akash Surendran , Krisztina Zsigmond , View ORCID Profile Ramón Alain Miranda-Quintana doi: https://doi.org/10.1101/2025.07.03.663077 Akash Surendran 1 Quantum Theory Project and Department of Chemistry, University of Florida , Gainesville, Florida 32611, United States Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Akash Surendran Krisztina Zsigmond 1 Quantum Theory Project and Department of Chemistry, University of Florida , Gainesville, Florida 32611, United States Find this author on Google Scholar Find this author on PubMed Search for this author on this site Ramón Alain Miranda-Quintana 1 Quantum Theory Project and Department of Chemistry, University of Florida , Gainesville, Florida 32611, United States Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Ramón Alain Miranda-Quintana For correspondence: quintana{at}chem.ufl.edu Abstract Full Text Info/History Metrics Preview PDF Abstract The visualization of high-dimensional chemical space is a critical tool for under-standing molecular diversity, structure–property relationships, and for guiding compound selection. However, the performance of non-linear dimensionality reduction (DR) techniques like t-Stochastic Neighborhood Embedding (t-SNE), Uniform Man-ifold Approximation and Projection (UMAP), and Generative Topographic Mapping (GTM) are often susceptible to the choice of hyperparameters, along with the high cost of their training for large datasets. In this study, we investigated the effect of undersampling methods on the choice of hyperparameter selection for these non-linear dimensionality reduction methods. Our results demonstrate that selecting small representative subsets of chemical data not only reduces computational costs associated with hyperparameter training but also serves as an innovative means to train non-linear DR methods, leading to projections that better preserve the local structure within the chemical space. Introduction The chemical space represents the multidimensional landscape of already existing as well as theoretically possible molecules, enclosing the “intermolecular relationships” based on the structural and functional properties of these molecules. 1 – 6 With advances in modern synthetic and computational fronts, this landscape is continuously expanding at a rapid rate, with estimates suggesting about 10 60 molecules residing in the drug-like chemical space. 5 These molecules are usually represented using vectors that span several hundreds or even thousands of dimensions, the simplest being binary fingerprints, where the presence and absence of a structural feature are represented by 1 and 0, respectively. 7 – 9 Similarity indices are usually employed to quantify the similarity between a pair of molecules, with the Tanimoto similarity 10 being the most widely used index for binary fingerprints. Mathematically, this function maps a pair of molecular fingerprints to a number between 0 and 1, with 0 and 1 representing completely dissimilar and identical molecules, respectively. The principle behind similarity-based searches is that “similar molecules display similar properties”. 11 Typically, the average similarity of a library is calculated using the set of all pairwise similarities, which scales as 𝒪( N 2 ) to the number of molecules. To address this issue, our group developed iSIM, 12 – 16 an extended similarity-based 17 – 21 tool to calculate the average similarity of molecular libraries in 𝒪( N ). The principle behind iSIM is the simultaneous comparison of multiple molecules, and it has been shown that the results are consistent with pairwise similarity across both binary fingerprints and real-valued descriptors. Coupled with complementary similarity, this method also supports advanced sampling of molecular libraries to target specific sectors of the vast chemical space, improving the coverage and representation of the selected compound sets. Following the identification of promising regions, a crucial next step is the visualization of the chemical space to interpret the complex structure-property relationships between molecules. Because the usual high-dimensional representation of molecules is not interpretable to the human eye, dimensionality reduction (DR) methods are usually employed to project these fingerprints to two or three dimensions. 22 – 24 Principal Component Analysis (PCA) 25 , 26 remains a workhorse technique for chemical space visualization, which works by identifying orthogonal axes of maximum variance in the descriptor space and projecting them along these axes, preserving the global data structure. However, this comes at the cost of poor local neighborhood preservation, especially when the chemical space is highly nonlinear. In a recent study by Orlov et al., 23 it was shown that nonlinear methods such as t-Stochastic Neighborhood Embedding (t-SNE), Uniform Manifold Approximation and Projection (UMAP), and Generative Topographic Mapping (GTM) provide far superior performance in preserving the local structure of the underlying chemical space, which mostly consists of small organic molecular compounds. Similarly, Takacs et al. 27 investigated the application of self-organizing maps (SOMs) to visualize and analyze regions of drug-like chemical space that remain heavily underrepresented. This method maps the highdimensional chemical space onto a two-dimensional grid, preserving the topological features and highlighting the property clusters. By transforming multidimensional chemical data into intuitive graphical representations, these methods can enhance the virtual screening of large molecular libraries, structure-activity relationships (SAR), library design, and synthesis of drug-like molecules. 4 , 5 , 28 Non-linear manifold-based dimensionality reduction methods on one hand provide key insights into the local distribution of data in molecular libraries but require extensive computational resources. The hyperparameter set plays a key role in the efficiency of these methods, and training these models on larger datasets can be challenging. In this study, we attempt to address this problem by training three non-linear models, t-SNE, UMAP, and GTM on 30 CHEMBL datasets 29 by implementing five different sampling methods using iSIM, but instead training the models on these smaller datasets. A comparison will be performed to determine whether the choice of sampling can help improve the quality of projections performed by these three DR methods. The performance of the projections will be evaluated on several local and global neighborhood preservation metrics 23 , 30 (discussed in the next section), and the results will be compared across different DR methods as well as within each DR method across different training samples used. The key question to be answered in this manuscript is thus: what is the impact of robust, deterministic, undersampling methods on the selection of training sets for non-linear DR techniques in chemical space visualization. Results and discussion Methodology and Workflow Generation of Fingerprints A brief visual representation of the workflow is provided in Figure 1 . CHEMBL datasets for 30 different macromolecular targets containing between 615 and 3657 molecules in SMILES strings were selected from the study by van Tilborg et al. 29 Binary Morgan fingerprints were generated using the RDKit module in Python. 8 , 31 In this representation, molecules are represented as fixed-length binary vectors, and the popular Extended-Connectivity Fingerprints (ECFP) with a radius of 2 and 1024 bit length, also known as the ECFP4 representation, was employed. Although the choice of molecular representation mainly depends on the focus of the study, it was recently shown that despite the simplicity of ECFP or RDKit fingerprints compared to more complex models such as Graph Neural Networks (GNNs) or pre-trained sequence-based transformers, these simpler models achieved a more robust performance for peptide property prediction even without hyperparameter tuning when evaluated over 126 datasets. 32 This suggests that long-range interactions (encoded by GNNs and transformers), previously thought to be critical, may not be essential for peptide function prediction, indicating that short-range features are sufficient. Hence, we are mainly interested in the local neighborhoods of molecules rather than the entire global structure of the dataset. Download figure Open in new tab Figure 1. Brief workflow: 30 CHEMBL datasets curated for 30 different macromolecular targets were selected for this study. Morgan fingerprints were generated for each dataset, and samples (10% and 20% of the original dataset size) were generated using iSIM complementary similarity sampling methods. Nonlinear DR methods were trained using each sample, and the corresponding original dataset was projected in 2D. Neighborhood Preservation Metrics were analyzed to compare the quality of these projections for chemical space visualization. iSIM complementary similarity based sampling of libraries iSIM is a novel computational method that efficiently calculates the average similarity between molecular sets in linear time scaling (𝒪( N )), bypassing the quadratic (𝒪( N 2 )) scaling of traditional pairwise approaches. It works by aggregating bit- or descriptor-level statistics across the entire set to calculate the average similarity of the entire chemical library. For example, to calculate the instant Tanimoto similarity ( iT ) for a set of N molecules represented by M ™dimensional binary fingerprints, the first step is to arrange all these vectors in a matrix with each row representing a molecule, that is, an N × M matrix. Then, a 1 × N matrix corresponding to the sum of the individual columns of the previous matrix is constructed, and each element of this matrix is represented by k q , q = 1, 2, …, M . Using this, iT can be calculated as After fingerprint generation, the datasets were sampled using complementary similarity- based sampling tools in the iSIM module. 12 The key idea behind complementary similarity is “how well a molecule represents others.” One molecule was removed, and the iSIM for the remaining set was calculated. A higher complementary similarity corresponds to low- density regions or lower similarity to the rest of the set, whereas the opposite is true for lower complementary similarity. Similarities were calculated using the Tanimoto similarity index, which is a widely used similarity index for binary data, and the molecules were ranked according to increasing complementary similarity. This can then be used to target specific regions of chemical space by providing a means to sample the set in 𝒪( N ) complexity. Using the tools in our group’s GitHub https://github.com/mqcomplab/iSIM/tree/main , we sampled 10 and 20% of each dataset by applying the following five sampling methods: Medoid sampling: M (10 or 20% of the total molecules in the set) Molecules with lowest complementary similarity Outlier sampling: M Molecules with highest complementary similarity Extremes sampling: M/2 molecules with highest complementary similarity and M/2 molecules with the lowest complementary similarity. Stratified sampling: Divides the chemical space into strata of equal size and selects the molecule with lowest similarity in each stratum. Quota sampling: The range of complementary similarity is divided into equal strata and molecules are selected to ensure the same complementary similarity range in each stratum. A visual representation of all five sampling methods is given in Figure 2 for CHEMBL234, the largest dataset in this study, which contains 3657 molecules. Medoid, outlier and extremes target specific regions of the chemical space, while stratified and quota samples target diverse regions of the underlying chemical space. Download figure Open in new tab Figure 2. t-SNE plots for CHEMBL234 by sampling 10% of the data. Blue portions depict the sampled molecules and grey portions represent the projection of the entire dataset. The former three methods target specific portions but the latter two target diverse regions of the chemical space. Hyperparameter tuning of nonlinear DR methods The next step involved hyperparameter tuning for the three nonlinear dimensionality reduction methods: t-SNE, UMAP, and GTM. DR models and metrics were chosen from the GitHub repository of the authors of. 23 The t-SNE algorithm was implemented using the OpenTSNE Python library, the UMAP algorithm using the umap-learn library, and the GTM using the Incremental GTM of Gaspar et al. 33 The set of hyperparameters was exactly the same as that in, 23 and a grid search was adopted to tune the best set of hyperparameters. For t-SNE, perplexity values were selected from [1, 2, 4, 8, 16, 32, 64, 128] and exaggeration from [1, 2, 3, 4, 5, 6, 8, 16, 32], learning rate as the default of OpenTSNE and Fast Fourier Transform accelerated interpolation was used for gradient calculation. For UMAP, the nearest neighbors (n neighbors) were chosen from [2, 4, 6, 8, 16, 32, 64, 128, 256], and the minimal distance (min dist) was chosen from [0.0, 0.1, 0.2, 0.3, 0.4, 0.6, 0.8, 0.99]. For GTM, the number of nodes was selected from [225, 625, 1600], the number of basis functions was set to [100, 400, 1225], the regularization coefficients were set to [1, 10, 100], and the basis widths were set to [0.1, 0.4, 0.8, 1.2]. The scoring was based on the percentage of preserved nearest 20 neighbors in the high-dimensional fingerprint space. Neighborhood Preservation Score In a pivotal contribution, Orlov et al. 23 recently highlighted the relevance of quantifying the degree to which different DR representations could retain information about local environments compared to the original high-dimensional fingerprint space. A combination of local and global preservation metrics was used to evaluate the efficiency of the neighborhood preservation. The first step involves finding the first k neighbors of a molecule in the “ambient” or 1024-dimensional fingerprint space based on the Tanimoto similarity. The same step is then repeated in the “latent” or low-dimensional projection space, but using Euclidean distances, since the Tanimoto distance is not a metric for real-valued vectors. 34 The number of nearest preserved neighbors can be evaluated using the neighborhood preservation score P NN ( k ): where k represents the number of neighbors considered and S ik represents the overlap or shared k-nearest neighbors of the N compounds in latent and ambient spaces. Additional local and global preservation metrics can be evaluated by constructing the co-ranking matrix Q, which is discussed in the next section. Co-ranking Matrix Q and Neighborhood preservation metrics A matrix can be constructed in both the ambient and latent spaces by evaluating and ranking the pairwise distances. This matrix is called the co-ranking matrix Q. The elements of the co-ranking matrix Q kl count the number of cases where samples of rank k in the ambient space become rank l in the latent space and are diagonal in the ideal case. This matrix can also be used to calculate the following DR metrics. 1 Co-k Nearest Neighbor size Represents the number of points within k-nearest neighbors before and after DR, giving the total number of mild intrusions and exclusions: where m represents the total number of neighbors, and k is the number of neighbors considered for the hit calculation. The factor 1 /km is a normalization factor that keeps the values in [0, 1] range. Here, mild intrusions refer to elements with k l and correspond to points that are pushed away by DR. Ideally, Q NN is 1, which means no change in the order of neighbors after DR. 2 Area under Q NN curve A single metric irrelevant of k to calculate the global neighborhood preservation based on the Q NN curve: 3 Local continuity meta criterion (LCMC) As an alternative to Q NN , LCMC removes the number of neighbors from Q NN : The metric favors local neighborhood preservation compared to Q NN by ensuring a larger penalty for a higher k value. k value corresponding to the maximum of LCMC ( k ), k max is 4 Local and Global Property metrics Calculated from the Q NN curve, Q local and Q global correspond to the local and global neighborhood preservation metrics. Given the k max value previously defined, these metrics can be calculated as follows: Q local is usually preferred over Q global because the immediate neighbors of a compound are much more significant in cheminformatics studies. 5. Trustworthiness and Continuity Trustworthiness and continuity are used to account the errors due to hard-intrusions and hard-extrusions respectively. The DR methods and neighborhood metrics were generated with the help of software from Varnek’s group and the code is available at https://github.com/AxelRolov/cdr_bench . Results The grid search was performed as follows: each of the 30 CHEMBL datasets was first sampled to obtain 10 and 20% of the data using the iSIM sampling tools corresponding to the five sampling methods previously discussed. These sampled sets were then used to perform hyperparameter tuning for the three nonlinear DR methods: t-SNE, UMAP, and GTM. Because these methods are susceptible to the choice of hyperparameters and the intrinsic structure of the data, there is no universal hyperparameter configuration, and the best selection of hyperparameters will always vary with the chosen datasets. For example, t-SNE was applied to published single-cell RNA-seq datasets to study the perplexity trade-offs between the local/global structure to show that although a high perplexity value (∼ 1% of sample size) can improve the global structure of the projected data, it comes at the cost of degraded local structures. 35 Despite the acceleration of gradient calculations provided by FFT and interpolation, t-SNE is still typically slower than PCA and UMAP (for the order of ∼ 10 4 datapoints), making the hyperparameter tuning process costly for large datasets. In our computation, the complete training and testing process of UMAP took the least time, 33s to train on 10% of the CHEMBL204 dataset (sample size = 274 molecules) and then project the entire dataset (containing 2754 molecules). On the other hand, GTM took the most time, about 12 minutes to do the same, followed closely by t-SNE which took about 11 minutes and 30s. In a previous work by the group, 12 it was shown that iSIM achieves linear scaling (𝒪( N )) in dataset sampling and when about 10% of the datasets were selected, quota and stratified sampling stand out in closely matching the average similarity of the entire dataset, achieving better coverage of the underlying chemical space. This is also clear from the visual representation of these sampling methods in Figure 2 . Though, the global data structure is better captured by quota and stratified samples, the main interest of this article is if the choice of sampling could influence the local structure after DR. A scatter plot of these three methods trained on the five sampling methods on one of the datasets used in the study (CHEMBL237) is shown in Figure 3 . The structure of projections shows significantly larger variations on t-SNE trained across different sampling methods to those for UMAP and GTM. This indicates that t-SNE is much more sensitive to the choice of hyperparameters than the other two DR methods. A larger portion of bright points in the GTM heatmap indicates superior local neighborhood preservation with the method trained on quota and stratified samples showing a larger number of bright points. Download figure Open in new tab Figure 3. t-SNE, UMAP and GTM projections of CHEMBL237 dataset trained on 20% samples by Extremes, Medoid, Outlier, Quota and Stratified sampling. The abundance of lighter shades in the heatmap of GTM indicates a superior local neighborhood preservation. t-SNE on the other hand, is more sensitive to the choice of sampling. To get a quantitative understanding of the quality of projections, the performance of these five sampling methods is evaluated based on the neighborhood metrics described in the previous section. The key idea is to sample datasets up to one or two orders of magnitude below the actual size, capturing as much information as possible about the distribution of the data, which further boosts the optimum hyperparameter selection at reduced computational costs. Considering the size of smaller datasets, the sample size was chosen to be at least 10%, to have a greater number of molecules than the minimum number of strata, which was set to 50 by default. The hyperparameter scoring is based on the percentage of the first 20 neighbors preserved in the latent space, according to the code developed for. 23 Because we are interested in the local neighborhood of a molecule rather than the global structure of the entire dataset, special focus is placed on metrics such as PNN ( k ), Q local , Trustworthiness and Continuity, as the presence of similar structural features or functional groups governs properties. A plot of these metrics against the number of neighbors k is given in Figure 4 Fig and are tabulated in Table 1 (the values correspond to k hit = 20). AUC and Q global describe the global neighborhood preservation quality of the DR method, while Q local , PNN, trustworthiness, and continuity focus on local neighborhood preservation. Although the performance of t-SNE and UMAP was consistent across all five sampling methods, there were key differences in the performance of GTM. First, while comparing the local structure preservation metrics across these three methods, it can be seen that GTM offers a far superior performance to t-SNE and UMAP, while the global metrics, such as AUC and Q global do not show a significant difference. The percentage of preserved nearest 20 neighbors (PNN) was less than 20 for t-SNE and UMAP, while the number was as high as 48% on average for GTM across all 30 CHEMBL datasets. This value was as high as over 50% when the first 50 neighbors were considered. This hints at the far superior performance of GTM over the other two methods while offering a similar global structure. View this table: View inline View popup Download powerpoint Table 1: Embedding quality of neighborhood preservation metrics across the five sampling methods for (a) t-SNE, (b) UMAP and (c) GTM. While the global preservation metrics do not show a large variation across DR methods, the local preservation is consistently better in GTM. GTM trained on quota and stratified samples show enhanced local metric preservation while the performance of t-SNE and UMAP are similar across all sampling methods. Download figure Open in new tab Figure 4. DR metrics plot in the order Extremes, Medoid, Outlier, Quota and Stratified. The color scheme is as follows: t-SNE: Yellow, UMAP: Green and GTM: Red. The shaded regions represent the standard deviation across datasets. Given the rapidly growing amount of chemical data, a central problem in cheminformatics is to train Machine Learning models with minimum input to reproduce maximum information about a given dataset. Hyperparameter selection is key for the performance of nonlinear DR methods and often computationally expensive given their non-linear scaling with data size, and hence, we address this problem by asking if a smaller sampling of these datasets could provide a suitable set of hyperparameters and hence a better quality of projections. The neighborhood preservation metrics were compared within each DR method, and it was observed that although these metrics remained consistent across all sampling methods for t-SNE and UMAP, there were stark differences in the quality of projections in GTM. While P NN shows an increasing trend with the number of neighbors k across all DR methods, GTM trained on quota and stratified sampling of datasets displayed an improved performance, preserving as high as over half of the nearest 50 neighbors in the latent space. Trustworthiness metric displays significantly better values in GTM, but more importantly clocks an average value of 93% when GTM is trained on Quota and Stratified sampling. Even for k = 50,the metric shows a value above 90% while other training samples show about 80% and an even lower number (between 60 and 70%) when for t-SNE or UMAP. A similar trend is also observed for the Continuity metric. Note that Trustworthiness and Continuity display the errors due to Hard intrusions and hard extrusions respectively. A high value for both metrics indicates that the DR method minimizes false positives (trustworthiness) and false negatives (continuity) in local relationships. LCMC , which is a local preservation-focused alternative to Q NN , peaks at a higher value for quota and stratified training samples, leading to a better Q local score, an important local preservation metric. Q local , in general shows a relatively better performance even for t-SNE and UMAP, although, the differences were more profound when comparison is within GTM, with average values as high as 0.45. A key reason for the overperformance of GTM compared to other DR methods is its probabilistic grid structure, which preserves the local topology better. Coupled with efficient iSIM sampling methods like quota and stratified sampling which are known to provide a better description of the data distribution, the choice of hyperparameter selection is enhanced. Even with a small fraction of data (approximately 10-20%), the set of hyperparameters corresponding to the intrinsic distribution in the original data set can be found at a highly reduced cost, leading to a better and more accurate low-dimensional projection of the data. Sampling methods covering diverse chemical regions outperform extremes-focused approaches. Another key observation is that stratified/quota sample trained projections lead Q global by a small margin (approximately ≤ 0.08) across methods, indicating that global structure is less sensitive to sampling than the local structure. A similar trend can also be observed in the case of AUC. It was also observed that sampling focused on outliers consistently ranked lower in local metrics: PNN, Q local , and Trustworthiness / Continuity, since prioritizing peripheral compounds could potentially sacrifice local structure integrity. Hence, the choice of samples chosen to train dimensionality reduction play a significant role in their performance, as was clearly observed in the case of advanced methods like GTM. Though, we did not observe a significant enhancement in the performance of t-SNE and UMAP in this study, we believe that a wider hyperparameter search could possibly help learn the intricate relationships in the distribution of data in chemical space, leading to better projections, which could be a topic of future research. Conclusions This work primarily focused on the effect of using small samples of the original dataset as the training data across three different dimensionality reduction methods: t-SNE, UMAP and GTM. It was observed that Quota and Stratified samples (size of 10% and 20% of the original dataset) used as training sets enhanced the performance over local neighborhood preservation metrics like PNN, Q local , Trustworthiness and Continuity, especially for GTM, while the effect of sampling was observed to be minimal while describing the global structure of data. So we conclude that sampling data from diverse regions of the chemical space helps in capturing intricate relationships in the manifold of chemical space, which is often overlooked when extreme-focused sampling approaches are used. Another important conclusion is that the choice of DR method plays a significant role in capturing these relationships, which directly is the reason for performance of GTM over t-SNE and UMAP. A study with a larger set of hyperparameters for t-SNE and UMAP could potentially improve their performance and this could be an interesting topic for future research. Acknowledgment A.S., K.Z. and R.A.M.Q. thank the National Institute of General Medical Sciences of the National Institutes of Health for support under award number R35GM150620. A.S thanks Alexey Orlov for valuable insights on the neighborhood preservation metrics code. Funder Information Declared National Institutes of Health, https://ror.org/01cwqze88 , R35GM150620 References (1). ↵ Lipinski , C. ; Hopkins , A. Navigating chemical space for biology and medicine . Nature 2004 , 432 , 855 – 861 . OpenUrl CrossRef PubMed Web of Science (2). ↵ Reymond , J.-L. The chemical space project . Accounts of chemical research 2015 , 48 , 722 – 730 . OpenUrl CrossRef PubMed (3). Reymond , J.-L. ; Ruddigkeit , L. ; Blum , L. ; Van Deursen , R. The enumeration of chemical space . Wiley Interdisciplinary Reviews: Computational Molecular Science 2012 , 2 , 717 – 733 . OpenUrl (4). ↵ Medina-Franco , J. L. ; Chóvez-Hernóndez , A. L. ; López-López , E. ; Saldívar-Gonzólez , F. I. Chemical multiverse: an expanded view of chemical space . Molecular Informatics 2022 , 41 , 2200116 . OpenUrl PubMed (5). ↵ Reymond , J.-L. ; Van Deursen , R. ; Blum , L. C. ; Ruddigkeit , L. Chemical space as a source for new drugs . MedChemComm 2010 , 1 , 30 – 38 . OpenUrl CrossRef Web of Science (6). ↵ Medina-Franco , J. L. ; Sónchez-Cruz , N. ; López-López , E. ; Díaz-Eufracio , B. I. Progress on open chemoinformatic tools for expanding and exploring the chemical space . Journal of Computer-Aided Molecular Design 2022 , 36 , 341 – 354 . OpenUrl CrossRef PubMed (7). ↵ David , L. ; Thakkar , A. ; Mercado , R. ; Engkvist , O. Molecular representations in AI-driven drug discovery: a review and practical guide . Journal of cheminformatics 2020 , 12 , 56 . OpenUrl PubMed (8). ↵ Rogers , D. ; Hahn , M. Extended-connectivity fingerprints . Journal of chemical information and modeling 2010 , 50 , 742 – 754 . OpenUrl CrossRef PubMed Web of Science (9). ↵ Landrum , G. ; others RDKit: Open-source cheminformatics . 2006 . (10). ↵ Bajusz , D. ; Rócz , A. ; Héberger , K. Why is Tanimoto index an appropriate choice for fingerprint-based similarity calculations? Journal of cheminformatics 2015 , 7 , 1 – 13 . OpenUrl CrossRef PubMed (11). ↵ Maggiora , G. ; Vogt , M. ; Stumpfe , D. ; Bajorath , J. Molecular similarity in medicinal chemistry: miniperspective . Journal of medicinal chemistry 2014 , 57 , 3186 – 3204 . OpenUrl CrossRef PubMed (12). ↵ López-Pérez , K. ; Kim , T. D. ; Miranda-Quintana , R. A. iSIM: instant similarity . Digital Discovery 2024 , 3 , 1160 – 1171 . OpenUrl PubMed (13). Lopez-Perez , K. ; Zhao , B. ; Miranda-Quintana , R. A. iSIM-Sigma: Efficient Standard Deviation Calculation for Molecular Similarity . Journal of Chemical Information and Modeling 2025 , (14). Lopez Perez , K. ; Lopez-Lopez , E. ; Soulage , F. ; Felix , E. ; Medina-Franco , J. L. ; Miranda-Quintana , R. A. Growth vs Diversity: A Time-Evolution Analysis of the Chemical Space . Journal of Chemical Information and Modeling 2025 , (15). López-Pérez , K. ; Miranda-Quintana , R. A. iCliff Taylor’s version: Robust and Efficient Activity Cliff Determination . Journal of chemical information and modeling 2025 , (16). ↵ López-Pérez , K. ; Miranda-Quintana , R. A. Extended Activity Cliffs-Driven Approaches on Data Splitting for the Study of Bioactivity Machine Learning Predictions . Molecular Informatics 2025 , 44 , e202400054 . OpenUrl PubMed (17). ↵ Miranda-Quintana , R. A. ; Bajusz , D. ; Rócz , A. ; Héberger , K. Extended similarity indices: the benefits of comparing more than two objects simultaneously. Part 1: Theory and characteristics . Journal of cheminformatics 2021 , 13 , 32 . OpenUrl PubMed (18). Miranda-Quintana , R. A. ; Rócz , A. ; Bajusz , D. ; Héberger , K. Extended similarity indices: the benefits of comparing more than two objects simultaneously. Part 2: speed, consistency, diversity selection . Journal of Cheminformatics 2021 , 13 , 33 . OpenUrl PubMed (19). López-Pérez , K. ; López-López , E. ; Medina-Franco , J. L. ; Miranda-Quintana , R. A. Sampling and mapping chemical space with extended similarity indices . Molecules 2023 , 28 , 6333 . OpenUrl PubMed (20). Dunn , T. B. ; López-López , E. ; Kim , T. D. ; Medina-Franco , J. L. ; Miranda-Quintana , R. A. Exploring activity landscapes with extended similarity: is Tanimoto enough? Molecular Informatics 2023 , 42 , 2300056 . OpenUrl (21). ↵ Chang , L. ; Perez , A. ; Miranda-Quintana , R. A. Improving the analysis of biological ensembles through extended similarity measures . Physical Chemistry Chemical Physics 2022 , 24 , 444 – 451 . OpenUrl CrossRef (22). ↵ Cihan Sorkun , M. ; Mullaj , D. ; Koelman , J. V. A. ; Er , S. ChemPlot, a Python library for chemical space visualization . Chemistry-Methods 2022 , 2 , e202200005 . OpenUrl (23). ↵ Orlov , A. A. ; Akhmetshin , T. N. ; Horvath , D. ; Marcou , G. ; Varnek , A. From high dimensions to human insight: Exploring dimensionality reduction for chemical space visualization . Molecular Informatics 2025 , 44 , e202400265 . OpenUrl PubMed (24). ↵ Reymond , J.-L. The chemical space project . Accounts of chemical research 2015 , 48 , 722 – 730 . OpenUrl CrossRef PubMed (25). ↵ Wenderski , T. A. ; Stratton , C. F. ; Bauer , R. A. ; Kopp , F. ; Tan , D. S. Principal component analysis as a tool for library design: a case study investigating natural products, brand-name drugs, natural product-like libraries, and drug-like libraries . Chemical Biology: Methods and Protocols 2015 , 225 – 242 . (26). ↵ Naveja , J. J. ; Medina-Franco , J. L. ChemMaps: Towards an approach for visualizing the chemical space based on adaptive satellite compounds . F1000Research 2017 , 6 , Chem – Inf . OpenUrl (27). ↵ Takócs , G. ; Sóndor , M. ; Szalai , Z. ; Kiss , R. ; Balogh , G. T. Analysis of the uncharted, druglike property space by self-organizing maps . Molecular Diversity 2021 , 1 – 15 . (28). ↵ Saldívar-Gonzólez , F. I. ; Medina-Franco , J. L. Approaches for enhancing the analysis of chemical space for drug discovery . Expert Opinion on Drug Discovery 2022 , 17 , 789 – 798 . OpenUrl PubMed (29). ↵ Van Tilborg , D. ; Alenicheva , A. ; Grisoni , F. Exposing the limitations of molecular machine learning with activity cliffs . Journal of chemical information and modeling 2022 , 62 , 5938 – 5951 . OpenUrl CrossRef PubMed (30). ↵ Zhang , Y. ; Shang , Q. ; Zhang , G. pyDRMetrics-A Python toolkit for dimensionality reduction quality assessment . Heliyon 2021 , 7 . (31). ↵ Landrum , g. Open-source cheminformatics software . https://www.rdkit.org/ . (32). ↵ Adamczyk , J. ; Ludynia , P. ; Czech , W. Molecular Fingerprints Are Strong Models for Peptide Function Prediction . arXiv preprint arxiv: 2501.17901 2025 , (33). ↵ Gaspar , H. A. ; Baskin , I. I. ; Marcou , G. ; Horvath , D. ; Varnek , A. Chemical data visualization and analysis with incremental generative topographic mapping: big data challenge . Journal of chemical information and modeling 2015 , 55 , 84 – 94 . OpenUrl PubMed (34). ↵ Surendran , A. ; Zsigmond , K. ; Lopez-Perez , K. ; Miranda-Quintana , R. A. Is the Tanimoto similarity a metric? Journal of Mathematical Chemistry 2025 , 63 , 1229 – 1240 . OpenUrl (35). ↵ Kobak , D. ; Berens , P. The art of using t-SNE for single-cell transcriptomics . Nature communications 2019 , 10 , 5416 . OpenUrl PubMed View the discussion thread. Back to top Previous Next Posted July 07, 2025. Download PDF Email Thank you for your interest in spreading the word about bioRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following Undersampling techniques for non-linear chemical space visualization Message Subject (Your Name) has forwarded a page to you from bioRxiv Message Body (Your Name) thought you would like to see this page from the bioRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share Undersampling techniques for non-linear chemical space visualization Akash Surendran , Krisztina Zsigmond , Ramón Alain Miranda-Quintana bioRxiv 2025.07.03.663077; doi: https://doi.org/10.1101/2025.07.03.663077 Share This Article: Copy Citation Tools Undersampling techniques for non-linear chemical space visualization Akash Surendran , Krisztina Zsigmond , Ramón Alain Miranda-Quintana bioRxiv 2025.07.03.663077; doi: https://doi.org/10.1101/2025.07.03.663077 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Bioinformatics Subject Areas All Articles Animal Behavior and Cognition (7636) Biochemistry (17703) Bioengineering (13897) Bioinformatics (41960) Biophysics (21458) Cancer Biology (18597) Cell Biology (25524) Clinical Trials (138) Developmental Biology (13381) Ecology (19905) Epidemiology (2067) Evolutionary Biology (24325) Genetics (15612) Genomics (22512) Immunology (17738) Microbiology (40421) Molecular Biology (17187) Neuroscience (88627) Paleontology (667) Pathology (2834) Pharmacology and Toxicology (4825) Physiology (7645) Plant Biology (15158) Scientific Communication and Education (2046) Synthetic Biology (4302) Systems Biology (9825) Zoology (2271)

Text is read by the "Ask this paper" AI Q&A widget below. Extraction quality varies by source — PMC NXML preserves structure cleanly, OA-HTML may include some navigation residue, and OA-PDF can have broken hyphenation. The publisher copy (via DOI) is the canonical version.

My notes (saved in your browser only)

Ask this paper AI returns verbatim quotes from the full text · source: preprint-html

Answers must be backed by verbatim quotes from this paper's full text. Hallucinated quotes are dropped automatically; if no verbatim passage answers the question, we say so. How this works

Citation neighborhood (no data yet)

We don't have any in-corpus citations linked to this paper yet. This is a recent paper (2025) — citers typically take a year or two to land, and the OpenAlex reference graph may still be filling in.

Source provenance

europepmc
last seen: 2026-05-20T01:45:00.602351+00:00