Regularized Deep Neural Networks for Combining Heterogeneous Features of Peptides in Data Independent Acquisition Mass Spectrometry

preprint OA: closed CC-BY-4.0
📄 Open PDF Full text JSON View at publisher

Abstract

Abstract Data-independent acquisition (DIA) has gained much attention in mass spectrometry (MS)-based proteomics for its improved reproducibility and unbiased data acquisition. In DIA-MS, the spectral library is crucial in peptide identification. However, this method is limited to peptides previously identified via data-dependent acquisition (DDA) MS experiments. This study proposes a deep learning approach for generating spectral libraries, even for previously unseen peptides. While most deep learning-based methods rely on one-hot encoding representation for peptides, the proposed method incorporates physicochemical features, including atomic composition, hydrophobicity, flexibility, fractional surface probability, and aromaticity. We introduce sparsity regu-larized neural network layers to facilitate the selection and combination of important high-dimensional physicochemical features and improve prediction performance. Fur-thermore, we suggest a transfer learning strategy for training the proposed deep neural networks having multiple heterogeneous input channels. Numerical experiments using benchmark DDA-MS data demonstrated that the proposed deep learning model out-performed existing benchmark models, such as Prosit and DeepDIA, particularly in predicting retention times. And it was demonstrated that the proposed models with sparsity regularization identified more peptides from HeLa cell DIA data compared to the other deep learning models.
Full text 55,732 characters · extracted from preprint-html · click to expand
Regularized Deep Neural Networks for Combining Heterogeneous Features of Peptides in Data Independent Acquisition Mass Spectrometry | bioRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-M677548'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search New Results Regularized Deep Neural Networks for Combining Heterogeneous Features of Peptides in Data Independent Acquisition Mass Spectrometry Namgil Lee , Hojin Yoo , Dohyun Han , View ORCID Profile Heejung Yang doi: https://doi.org/10.1101/2025.06.09.658564 Namgil Lee † Bionsight, Inc. , Gangwondaehak-gil 1, Chuncheon, Gangwon 24341, Republic of Korea ‡ Department of Information Statistics, Kangwon National University , Chuncheon, Gangwon 24341, Republic of Korea Find this author on Google Scholar Find this author on PubMed Search for this author on this site Hojin Yoo † Bionsight, Inc. , Gangwondaehak-gil 1, Chuncheon, Gangwon 24341, Republic of Korea Find this author on Google Scholar Find this author on PubMed Search for this author on this site Dohyun Han ¶ Department of Biomedical Sciences, Seoul National University College of Medicine , Seoul, Republic of Korea Find this author on Google Scholar Find this author on PubMed Search for this author on this site Heejung Yang † Bionsight, Inc. , Gangwondaehak-gil 1, Chuncheon, Gangwon 24341, Republic of Korea § Department of Pharmacy, Kangwon National University , Gangwondaehak-gil 1, Chuncheon, Gangwon 24341, Republic of Korea Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Heejung Yang For correspondence: heejyang{at}kangwon.ac.kr Abstract Full Text Info/History Metrics Preview PDF Abstract Data-independent acquisition (DIA) has gained much attention in mass spectrometry (MS)-based proteomics for its improved reproducibility and unbiased data acquisition. In DIA-MS, the spectral library is crucial in peptide identification. However, this method is limited to peptides previously identified via data-dependent acquisition (DDA) MS experiments. This study proposes a deep learning approach for generating spectral libraries, even for previously unseen peptides. While most deep learning-based methods rely on one-hot encoding representation for peptides, the proposed method incorporates physicochemical features, including atomic composition, hydrophobicity, flexibility, fractional surface probability, and aromaticity. We introduce sparsity regu-larized neural network layers to facilitate the selection and combination of important high-dimensional physicochemical features and improve prediction performance. Fur-thermore, we suggest a transfer learning strategy for training the proposed deep neural networks having multiple heterogeneous input channels. Numerical experiments using benchmark DDA-MS data demonstrated that the proposed deep learning model out-performed existing benchmark models, such as Prosit and DeepDIA, particularly in predicting retention times. And it was demonstrated that the proposed models with sparsity regularization identified more peptides from HeLa cell DIA data compared to the other deep learning models. Introduction Mass spectrometry (MS) has become a standard tool in proteomic research for the proteomic analyses of biological systems 1 , 2 . In liquid chromatography-tandem MS (LC-MS/MS), a protease digests proteins into peptides, which are further fragmented into fragment ions. The relative abundance of the fragment ions is then measured in the form of mass spectra. One of the most widely used data acquisition strategies for tandem MS is data-dependent acquisition (DDA) strategy: only a subset of precursor ions selected from the MS survey scan is processed in subsequent MS scans to obtain tandem MS spectra. In contrast, a data-independent acquisition (DIA) strategy has been introduced more recently as an accurate and reproducible method for detecting large fractions of peptides 3 . The DIA strategy collects tandem MS spectra from a series of MS scans covering broad precursor ion ranges, guaranteeing unbiased and consistent data collection. For peptide identification and quantification, a DIA strategy requires fragment ion spectral libraries generated for target peptides. The fragment ion intensities obtained from current experimental data are correlated with a spectral library for peptide identification 4 . Hence, the quality of spectral libraries is crucial for accurate peptide identification and quantification in DIA MS. Traditional methods for generating spectral libraries depend on fragment ion spectra obtained from the DDA data. Because of this dependence, the performance of the generated spectral libraries can be hindered by a mismatch in the instrument type, limited coverage of DDA samples, or high false discovery rates 5 . To overcome the limitations of traditional proteomics approaches based on MS, researchers have developed deep learning methods for various aspects of the proteomics workflow. Deep learning methods for predicting fragment ion intensities and retention times have been developed and applied to in silico spectral library generation 6 . An advantage of deep learning, and more generally machine learning, is that the inferred models can be generalized to new peptides. For example, peptides originating from another organism can be predicted 7 . However, traditional machine learning methods use a limited number of known features for peptide representation. For instance, each peptide is represented by a 20-dimensional vector consisting of 20 amino acid residues 8 . Because unknown factors and structural information on peptides can affect the measured spectra, using only known features can cause high prediction errors. In contrast, modern deep learning methods, such as convolutional neural networks (CNN) and recurrent neural networks (RNN) 9 , have been adopted to build models that can detect peptide features automatically. Because of the sequence structure of peptides, most popular deep learning methods are based on neural networks originally developed for natural language processing, such as the RNN 10 , long short-term memory (LSTM) 11 , and gated recurrent unit (GRU) 12 , 13 . LSTM and GRU are suitable for modeling short- and long-range interactions between the amino acids in peptides. Many deep learning methods for fragment ion intensity (MS2) prediction have been developed based on bidirectional LSTM (or GRU) and onehot encoding representations for peptides 14 – 19 . Using a one-hot encoding representation, scientists have encoded each amino acid as a 20-dimensional vector of zero-one values. Deep learning models automatically extract latent features via gradient-based optimization of model parameters. However, the one-hot encoding representation of peptides has potential drawbacks. First, many amino acid modifications exist; thus one-hot encoding representation cannot be generalized to potentially unseen modified peptides. Second, it may not produce chemically or physically interpretable latent features in deep learning models. Third, the performance of prediction models depends heavily on the data quality and computational power, which ignores prior knowledge of the chemical properties of peptides. Alternatively, a CNN-based deep learning method, MS2CNN, converts peptide sequences into feature vectors of peptide properties, such as amino acid composition, secondary structure, and physicochemical features 20 . In addition, to improve the generalization ability of deep learning models to previously unseen modified peptides, Bouwmeester et al. 21 proposed using peptide features based on atomic composition. Although all the studies mentioned above have proposed featurization methods for peptides and amino acids, combining various features to achieve high performance and physiological interpretation is worthwhile. This paper proposes a regularization method to combine one-hot encoding features with high-dimensional physicochemical features of peptides. The advantages of the proposed method are: first, using one-hot encoding features can enable deep learning models to extract latent features from data automatically. Second, the generalization performance of the deep learning model can be dramatically improved for modified peptides not included in the training data using peptide features based on atomic composition. Third, sparsity regularization prevents overfitting and selects useful features for MS spectral prediction, which improves interpretability of model performances. Fourth, high-dimensional heterogeneous features can be combined efficiently by a transfer learning strategy for training deep neural networks having multiple input channels. Figure 1 illustrates the proposed in silico spectral library generation method based on sparsity regularized deep neural networks. Download figure Open in new tab Figure 1 : Introduction to the proposed sparsity regularized deep neural networks for generating in silico spectral libraries. Methods Two deep learning models are required separately to generate a deep learning based spectral library: one for predicting fragment ion intensities and the other for predicting indexed retention time (iRT). In this work, both models are constructed consisting of featurization layers, feature extraction layers, a concatenation layer, and a fully connected layer, as depicted in Figure 2 . A more detailed description is provided below: Download figure Open in new tab Figure 2: Architecture of the proposed deep learning models Heterogeneous Features of Peptides The proposed deep learning models use modified peptide sequence as input. The featurization layers consist of multiple parallel converters for extracting heterogeneous numerical features from the peptides. The features can be grouped into four categories: each is forwarded to a proper deep neural network pipeline in the feature extraction layers to extract higher-level features at deeper layers. One-hot Encoding Features One-hot encoding is a featurization technique motivated by natural language processing. A sentence is represented as a sequence of words; likewise, a peptide can be represented as a sequence of amino acids. Considering the dictionary of the 20 standard amino acids, onehot encoding of a peptide converts each amino acid in the peptide into a zero-one-valued vector of length 20, where zeros and ones indicate the existence of a specific amino acid. In addition, because peptide sequences vary in length, the length difference is adjusted to fixed length through zero padding so that an RNN-based deep learning model can be applied in any subsequent layer. Consequently, the one-hot encoding of a peptide is represented as a zero-one matrix of size L × A , where L is the zero-padded peptide length and A is the number of amino acids in the dictionary. Although one-hot encoding has been widely adopted in deep learning approaches, it has limitations in generalizing to a wide variety of peptide modifications. Combining physicochemical features with one-hot encoding features can improve the generalization performance of deep learning models. Amino Acid Scale Features The amino acid scale features are the physicochemical properties of each amino acid, such as hydrophobicity 22 . While there are many predefined scales introduced in the literature, we chose five representative scale features to be incorporated in the numerical experiments: (1) the Kyte & Doolittle index of hydrophobicity 23 , (2) Wilson hydrophobic constants 24 , (3) flexibility parameters 25 , (4) fractional surface probability 26 , and (5) interior-to-surface transfer energy scale 27 . A peptide’s amino acid scale features are represented as a matrix of size L × S , where L is the zero-padded peptide length and S is the number of chosen scale features. Atom Count Features Atom count features represent the atomic composition of each amino acid. Basic atom count features were used to calculate the numbers of five atoms (C, H, N, O, and S) in each of the 20 standard amino acid types. The numbers of added and deleted atoms were calculated separately and integrated into the total number of atoms for modified amino acids. Consequently, the atom count features of a modified peptide are a matrix of size L × 5. Computing the atom count features is straightforward for any peptide; thus, any model using atom count features can be generalized to a modified amino acid with a known atomic composition. Global Features Global features encode the physicochemical properties of a peptide 22 . We retrieved 16 global features of a peptide using the SeqUtils module in the Biopython package 28 : length, molecular weight, aromaticity, charge at pH 5.0, charge at pH 7.5, minimum flexibility, median flexibility, maximum flexibility, grand average of hydropathy 23 , instability index 29 , isoelectric point 30 , molar extinction coefficient assuming cysteine (reduced), molar extinction coefficient assuming cysteine residues (Cys-Cys bond), the fraction of amino acids in the helix, the fraction of amino acids in the turn, and the fraction of amino acids in the sheet. The global features of a peptide are represented with a vector of length 16. Regularized Neural Networks for Heterogeneous Data Fusion Multiple heterogeneous features obtained in the featurization layers are forwarded to the deep neural network layers for higher-level feature extraction. Specifically, pipelines of deep neural network layers exist for processing each type of heterogeneous feature separately. The pipelines of deep neural network layers can be categorized into two types according to the input feature size. One type is for matrix-shaped features and the other is for vector-shaped features. Matrix-shaped features can encode sequence information of the peptide by preserving the positions of every amino acid in the sequence. In this case, the deep neural network layers consist of 1D CNN layers and bidirectional long short-term memory (BiLSTM) layers motivated by DeepDIA 19 . The kernel sizes of the 1D CNNs were set to two such that the CNN kernels could extract local feature patterns of the two neighboring amino acids from the whole sequence structure 9 . Conversely, vector-shaped features can encode the global properties of peptides rather than sequence-structure information. The deep neural network layers for vector-shaped features consist of fully connected layers with a rectified linear unit (ReLU) as the activation function. With dropout layers placed between the fully connected layers, the resulting deep neural network layers form a multilayered perceptron representing a nonlinear transformation of the input features. Notably, the input features are heterogeneous and high-dimensional; thus, redundant features may have been included. L1-norm regularization is known to prevent overfitting and to yield sparse weights 31 . Hence, for each pipeline of deep neural network layers described above, the weights of the first layer were regularized by the L1-norm, except the pipeline for one-hot encoding features. In other words, let W and b denote the kernel weights and bias of the first CNN layer or the first fully connected layer in a deep neural network pipeline. Then, during the training of the model, an optimizer minimizes the L1-norm regularized loss function, which can be expressed as where L ( W, b ) is a loss function, λ ≥ 0 is a regularization parameter, and ∥ W ∥ 1 = ∑ i | w i | is the L1-norm of weight W = ( w i ). The regularization parameter λ can be set differently for each deep neural network pipeline such that each type of feature can be weighted adaptively. Transfer Learning Because the proposed model consists of multiple deep neural network pipelines corresponding to heterogeneous input channels, gradient-based learning algorithms suffer from local optima and poor weight initialization. To mitigate such problems, we suggest a transfer learning strategy, which is depicted in Figure 2 . First, a deep neural network consisting of a featurization layer, a feature extraction layer, and a fully connected layer is constructed, where the featurization layer extracts only one-hot encoding features from peptides. Second, hyperparameters of the deep neural network are optimized and the network weights are trained. Third, the weights of the feature extraction layer are fixed and transferred to an another deep learning model, called Fusion, consisting of multiple deep neural network pipelines. Finally, hyperparameters of the Fusion model are optimized and the network weights are trained, except the weights of the transferred feature extraction layer corresponding to the one-hot encoding features. Hyperparameter Optimization We compared several deep learning models to generate in silico spectral libraries. The deep learning models used in this study can be categorized into (1) MS2 prediction models and (2) iRT prediction models. Note that the MS2 prediction models were trained separately on the precursor charges of +2 and +3 values. A charge 2 model and a charge 3 model may have different sets of hyperparameters even if they consist of the same types of deep learning layers. The names and brief description of the models are as follows. (A) Prosit: Pretrained Prosit models 15 , 32 . MS2: A Prosit model released in 2020 with higher energy C trap dissociation (HCD) fragmentation 32 . iRT: A Prosit model released in 2019 15 . (B) Baseline: Deep learning models provided by DeepDIA 19 . MS2: The input features consisted of only one-hot encoding features, and the model consisted of a single 1D-CNN layer, a single BiLSTM layer, and a single fully connected layer. iRT: The model used one-hot encoding features as input and consisted of a 1DCNN layer, a max-pooling layer, a BiLSTM layer, and three fully connected layers. (C) CRNN: Extension to the DeepDIA. MS2: The input features consisted of only one-hot encoding features, and the model consisted of multiple 1D-CNN layers, multiple BiLSTM layers, and multiple fully connected layers. iRT: Extension to the DeepDIA. The model consisted of multiple 1D-CNN layers, a max pooling layer, multiple BiLSTM layers, and multiple fully connected layers. (D) Fusion: The proposed deep learning model using multiple deep neural network pipelines corresponding to heterogeneous input channels. As explained in Section Transfer Learning, the weights of the feature extraction layer of the CRNN model were transferred to the pipeline for the one-hot encoding features. The first layer of each of the other pipelines was L1-norm regularized. (E) Fusion w.o. L1: The Fusion model without L1-norm regularization. (F) Fusion w.o. scale: The Fusion model without amino acid scale features. For hyperparameter optimization, the entire training set was split into training and validation sets with split ratios of 67% and 33%, respectively. The best set of hyperparameters was selected at the minimum of the validation set losses: the mean squared error (MSE) for the MS2 prediction models and the mean absolute error (MAE) for the iRT prediction models. See Supporting Information for detailed model architectures. Hyperparameter optimization and numerical experiments were performed on a workstation with an Intel CPU (3.00 GHz) and an NVIDIA GPU (GeForce RTX 3060 Ti) running Ubuntu 18.04. All deep learning models were implemented in Python 3.9 and Tensorflow 2.4. Hyperparameter optimization was performed using the Comet machine learning platform ( https://www.comet.com/ ) with the Python package comet-ml 3.1. Results Data Preparation The LC-MS/MS DDA data from four experiments were used to generate experimental spectral libraries. Each of these libraries was subsequently used to train deep learning models, with the HeLa data specifically utilized for hyperparameter optimization. A summary of the generated spectral libraries are provided in Table 1 . View this table: View inline View popup Download powerpoint Table 1: Summary of the experimental spectral libraries used for training the deep learning models. First, the DDA data from HeLa cells were acquired on a Q Exactive HF mass spectrometer (Thermo Fisher Scientific, Waltham, MA, USA), available at ProteomeXchange ( proteomecentral.proteomexchange.org/ ), with the dataset identifier PXD005573 33 . The 17 raw file names begin with C_D160304_S251-Hela-2ug-2h_MSG, C_D160331_S209-HPRP-HeLa, and C_D160401_S209-HPRP-HeLa. The DDA data were analyzed using Spectronaut 15 34 with default settings, except that Carbamidomethyl (C) was selected as the fixed modification, and no other modifications were applied. The DDA data were searched against the UniProt Homo sapiens FASTA database ( https://www.uniprot.org/ , accessed on 2021-09-13, containing 20,588 entries). In addition, LC-MS/MS DIA data for HeLa cells were obtained from the same repository of ProteomeXchange with the dataset identifier PXD005573. The DIA data were technical replicates acquired using the same experimental settings, which might be useful for validating the reproducibility of peptide identification. The three raw file names begin with Fig1_MP-DIA-120min120kMS1-22W30k-8dppp. Second, the experimental spectral library generated from Mouse cerebellum experiments was obtained directly from ProteomeXchange with the dataset identifier PXD014108. The original raw DDA data files were acquired on a Q Exactive HF mass spectrometer from 15 fractionated runs and are available at ProteomeXchange with the dataset identifier PXD005573. The raw files were searched using SpectroMine (version 1.0.21621) with the default settings, against the SwissProt Mus musculus FASTA database (accessed on 2019-02, containing 17,006 entries). The generated spectral library is available in the zipped file mouse_cerebellum_DDA.csv.zip. All cysteine residues in the peptide sequence were considered modified by carbamidomethylation. See Yang et al. 19 for more details on the database searching of the “Mouse2” DDA data. Lastly, the DDA data files from S. pombe cells were acquired on an Orbitrap Elite mass spectrometer (Thermo Fisher Scientific, Waltham, MA, USA) and are available at ProteomeXchange with the dataset identifier PXD011459. The S. pombe cell lysates were processed either with chymotrypsin alone, or with trypsin followed by chymotrypsin. The acquired DDA data files were searched against the PombeBase FASTA database ( https://www.pombase.org/ , accessed on 2024-01, containing 47,432 entries) using MaxQuant (version 2.4.10.0) 35 . The MaxQuant parameters were set as follows: Carbamidomethyl (C) as a fixed modification, Oxidation (M) as a variable modification, MS accuracy of 4.5 ppm, MS/MS tolerance of 0.5 Da, and a maximum number of missed cleavages set to 6. See Dau et al. 36 for more details. Numerical Evaluation The deep learning models were trained and tested using the experimental spectral libraries described in the previous section. Precursors with charge states greater than 3 or less than 2 were removed for training the deep learning models because their number was much smaller than that of the other precursors, which could lead to highly biased training and validation sets. To maintain the data split ratio identical to that of the hyperparameter optimization step, we split the entire training set into a training set and a validation set with a 2:1 ratio. The models were then numerically compared based on the validation set. HeLa Data Table 2 summarizes the performance of deep learning models in terms of the MSE, MAE, and mean absolute percentage error (MAPE). The Fusion w.o. L1 and CRNN models yielded the lowest MSE and MAE values among the MS2(+2) prediction models, whereas the Baseline model showed overall low performance because its architecture had not been optimized. The CRNN model yielded the lowest MSE and MAE values among the MS2(+3) prediction models, which implies that a relatively simpler model was sufficient for predicting the MS2 of precursors with charge state +3 than with charge state +2. On the other hand, the Baseline and Fusion models yielded the lowest MAPE values, which is due to their relatively simple architectures and L1-norm regularization that enhanced robustness to outliers. The Fusion w.o. L1 model achieved the lowest MSE and MAE values among the iRT prediction models, demonstrating that physicochemical features offer a significant advantage for iRT prediction. The pretrained Prosit models yielded the worst performance because they had been trained on different datasets. View this table: View inline View popup Download powerpoint Table 2: Comparison of in silico spectral library generation models trained and validated on HeLa data. The MSE, MAE, and MAPE are reported with standard errors listed below each one. The bold font indicates the smallest number among the models. Moreover, we selected the set of modified peptides among the validation set to evaluate the prediction performance for modified peptides. Table 3 summarizes the validation set errors of the deep learning models on modified peptides. Comparing Table 2 and 3, modified peptides were more difficult to predict and yielded larger errors than unmodified peptides for MS2 prediction; however, this phenomenon was reversed for iRT prediction. Notably, the Fusion w.o. L1 model achieved the best performance consistently for iRT prediction. This suggests that incorporating physicochemical features into the model helps improve prediction performance. View this table: View inline View popup Download powerpoint Table 3: Comparison of in silico spectral library generation models trained and validated on HeLa data, where prediction errors were computed only for unseen modified peptides in validation sets. The MSE, MAE, and MAPE are reported with standard errors listed below each one. The bold font indicates the smallest number among the methods. Mouse Cerebellum Data Table 4 presents the prediction performance of deep neural network models evaluated on the Mouse cerebellum data. Similar to the results for the HeLa data, the Fusion w.o. L1 and CRNN models achieved the lowest MSE and MAE values, while the Fusion model yielded the lowest MAPE values among MS2 prediction models. This result demonstrates the overall high accuracy of the Fusion w.o. L1 and CRNN models, as well as the robustness provided by the L1-norm regularization in the Fusion model. For iRT prediction, on the other hand, the Fusion w.o. L1 and Fusion w.o. scale achieved the best performance, suggesting that the proposed fusion models using atom count features and global physicochemical features are well-suitable for iRT prediction. View this table: View inline View popup Download powerpoint Table 4: Comparison of in silico spectral library generation models trained and validated on Mouse cerebellum data. The MSE, MAE, and MAPE are reported with standard errors listed below each one. The bold font indicates the smallest number among the models. Pombe Data Processed with Chymotrypsin Table 5 presents the prediction performance of the deep neural network models evaluated on the Pombe (chymotrypsin) data. In the table, the Fusion w.o. L1 and CRNN models achieved the lowest MSE and MAE values in MS2(+2) prediction, which is consistent with the previous results. Similarly, in MS2(+3) prediction, the Fusion w.o. L1 model achieved the lowest MSE value and the CRNN model achieved the lowest MAE and MAPE values. The Fusion w.o. L1 model consistently achieved the highest performance in iRT prediction, demonstrating the critical role of physicochemical features in accurate iRT estimation. View this table: View inline View popup Download powerpoint Table 5: Comparison of in silico spectral library generation models trained and validated on Pombe (chymotrypsin) data. The MSE, MAE, and MAPE are reported with standard errors listed below each one. The bold font indicates the smallest number among the models. Table 6 presents a comparison of the performance of deep neural network models on modified peptides from the Combe (chymotrypsin) dataset. Similar to Table 5 , the Fusion w.o. L1 model achieved the lowest MSE and MAE values in MS2(+2) prediction, while the CRNN obtained the lowest MAE and MAPE values in MS2(+3) prediction. Additionally, the Fusion w.o. L1 model outperformed others by yielding the lowest MSE, MAE, and MAPE values in iRT prediction. Notably, the Baseline model achieved the lowest MAPE value in MS2(+2) prediction and the lowest MSE value in MS2(+3) prediction. This suggests that its simple model architecture may have reduced overfitting in small datasets, while the incorporation of physicochemical features in the Fusion w.o. L1 model helped prevent overfitting and improved generalization. View this table: View inline View popup Download powerpoint Table 6: Comparison of in silico spectral library generation models trained and validated on Pombe (chymotrypsin) data, where prediction errors were computed only for unseen modified peptides in validation sets. The MSE, MAE, and MAPE are reported with standard errors listed below each one. The bold font indicates the smallest number among the methods. Pombe Data Processed with Trypsin and Chymotrypsin Tables 7 and 8 present the performance of deep neural network models on the Pombe (trypsinchymotrypsin) dataset. This dataset is slightly larger than the Pombe (chymotrypsin) dataset, as sequential digestion with two proteases improved peptide and protein identification 36 . While the Fusion w.o. L1 model achieved the lowest MSE, MAE, and MAPE in iRT prediction, it is notable that the Fusion w.o. scale model showed the second-best performance. This suggests that physicochemical features other than amino acid scale features– such as atom count and global features–also contributed to improved iRT prediction. View this table: View inline View popup Download powerpoint Table 7: Comparison of in silico spectral library generation models trained and validated on Pombe (trypsin-chymotrypsin) data. The MSE, MAE, and MAPE are reported with standard errors listed below each one. The bold font indicates the smallest number among the models. View this table: View inline View popup Download powerpoint Table 8: Comparison of in silico spectral library generation models trained and validated on Pombe (trypsin-chymotrypsin) data, where prediction errors were computed only for unseen modified peptides in validation sets. The MSE, MAE, and MAPE are reported with standard errors listed below each one. The bold font indicates the smallest number among the methods. Comparison of Spectral Libraries Generated by Deep Learning Models This section compares the spectral libraries generated by the deep learning models. An in silico spectral library can be generated by combining an MS2 prediction model and an iRT prediction model. We combined the same types of the MS2 and iRT prediction models as listed in Section Hyperparameter Optimization. In this study, we prepared two separate lists of peptides to predict MS2 and iRT values. One is the peptide list from the Pan-Human spectral library 37 , and the other is the union of the peptide lists from the Pan-Human spectral library and HeLa cell-based spectral library, which has been mentioned in previous paragraphs. We can see that the in silico spectral libraries can predict spectra of the peptides that have not been included in the HeLa-cell based experimental library. Table 9 compares the spectral libraries generated by deep learning models. Each row of the table represents quantitative metrics for the comparison. The spectral libraries were compared in terms of the number of unique peptide precursors including charge (Precursors), the number of unique stripped sequences (Peptides), the number of unique stripped sequences which point to only one protein (Proteotypic Peptides), the number of unique protein groups identified by a single peptide precursor (Single Hits), the number of fragment ions (Fragments), and the total number of fragment ions (Fragments Total). In Table 9 , we can find that the Fusion w.o. L1 models yielded the largest numbers of peptides and fragments, and the Fusion model yielded the least numbers. This result implies that neural network weights suppressed by the L1-norm regularization affected the reduced number of fragment ions. View this table: View inline View popup Download powerpoint Table 9: Comparison of spectral libraries generated by deep learning models regarding the number of peptides, proteins, and fragments included. The HeLa-cell based experimental spectral library used as the training dataset is listed in the right-most column for comparison. The spectral libraries generated by the deep learning models were imported and filtered using Spectronaut based on default settings. The quantitative metrics in the table rows are unique to MS2 prediction models. DIA Data Analysis Performance Comparison In this section, we compare in silico-generated spectral libraries in terms of peptide and protein identification using benchmark DIA datasets. Raw DIA datasets and generated in silico spectral libraries were imported and analyzed using Spectronaut 16 with default settings. Based on deep learning models trained on the DDA data of HeLa cells, two types of spectral libraries were generated: one is for the peptides from the Pan-Human spectral library and the other for peptides from the Pan-Human and HeLa spectral libraries. The upper panel of Table 10 summarizes the numbers of identified peptides and proteins when the spectral libraries were generated for the peptides contained in the Pan-Human spectral library. Among the deep learning methods compared, the CRNN model yielded the greatest numbers of identified peptides, and the Fusion w.o. scale model yielded the greatest numbers of proteins. Notably, the Fusion model yielded the least numbers of identified peptides and proteins, implying that regularizing neural networks for amino acid scale features might have adversely affected the identification of peptides and proteins. And the greatest numbers of identified proteins yielded by the Fusion w.o. scale model demonstrates that the other physicochemical features processed by regularized neural networks were helpful for improved identification of proteins. View this table: View inline View popup Download powerpoint Table 10: Comparison of spectral libraries regarding the numbers of identified peptides and proteins from HeLa cell DIA data. In this experiment, deep learning models were used to generate in silico spectral libraries for the peptides contained either in the Pan-Human spectral library or in the Pan-Human and HeLa cell-based experimental spectral library. The bold font indicates the largest number among different models. The lower panel of Table 10 presents the numbers of identified peptides and proteins obtained by in silico spectral libraries generated for peptides contained in the union of the Pan-Human and HeLa experimental spectral libraries. Comparing the upper and lower panels, we can find that including the peptides in the HeLa experimental spectral library increased the numbers of identified peptides and proteins significantly. Comparing the deep learning models, we can find that the Fusion w.o. L1 model yielded the largest numbers of identified peptides and proteins, while the Fusion model yielded the smallest numbers. This implies that while the L1-norm regularization on physicochemical features might have been helpful in the identification of proteins in the Pan-Human spectral library, it adversely affected in the HeLa experimental spectral library. Figure 3 illustrates the numbers of identified peptides and protein groups by the in silico spectral libraries generated by the Baseline, CRNN, and Fusion w.o. scale models, where the peptide list of the libraries was retrieved from the Pan-Human spectral library. The results from the other models were omitted for simplicity. The lower panels of the figure show that a large number of peptides are shared by the CRNN and Fusion w.o. scale models and a large number of proteins are shared by the Baseline and Fusion w.o. scale model. Moreover, it is shown that the Fusion w.o. scale model identified the largest number of protein groups, demonstrating the competitiveness of the proposed model. Download figure Open in new tab Figure 3 : Number of identified peptides (left panels) and protein groups (right panels) from HeLa cell DIA data, where in silico spectral libraries were generated for the peptides contained in the Pan-Human spectral library. Figure 4 illustrates the numbers of identified peptides and protein groups by the in silico spectral libraries generated by the Baseline, CRNN, and Fusion w.o. L1 models, where the peptide list of the libraries was retrieved from the Pan-Human and HeLa experimental libraries. The Fusion w.o. L1 model was selected because it showed the largest numbers of identified peptides and proteins in the lower panel of Table 10 . The lower panels of the figure show that more peptides and protein groups are shared by the CRNN and Fusion w.o. L1 models, and the Fusion w.o. L1 model identified the largest number of protein groups. Download figure Open in new tab Figure 4 : Number of identified peptides (left panels) and proteins (right panels) from HeLa cell DIA data, where in silico spectral libraries were generated for the peptides contained in the Pan-Human and the HeLa experimental spectral libraries. Notably, relatively large numbers of peptides were identified uniquely by the Baseline model in Figures 3 and 4 , however, the Fusion w.o. scale and Fusion w.o. L1 models identified larger numbers of proteins uniquely. This implies that the Fusion models could identify more peptides which were important for identifying proteins than the Baseline model. Discussion and Conclusions This study proposes a deep learning-based method for in silico spectral library generation. The proposed method effectively combines traditional one-hot encoding features with the physicochemical features of peptides by employing L1-norm regularization and a transfer learning strategy. The physicochemical features used in this study are categorized into vectorvalued and matrix-valued types. Vector-valued features represent the global features of the peptide as a whole, whereas matrix-valued features capture characteristics extracted based on the sequence of amino acids constituting the peptide. The regularized neural networks used in this study are designed to process both types of features in parallel and integrate them through fully connected layers. Furthermore, the proposed transfer learning strategy improves the training efficiency by dividing a high-dimensional optimization problem into two separate, lower dimensional problems. In numerical experiments using data obtained from cell lines of different species, including human, mouse, and S. pombe, the proposed method outperformed existing deep learning methods such as Prosit and DeepDIA. Specifically, the results showed that hyperparameter optimization led to well-tuned deep neural network architectures, resulting in superior performance by the CRNN and the family of Fusion models. Furthermore, comparison with model variants–featuring slightly different combinations of input features and hyperparameters–demonstrated that the Fusion w.o. L1 and CRNN models achieved the lowest MSE and MAE values in most MS2 prediction tasks, and the Fusion w.o. L1 model consistently showed the lowest MSE, MAE, and MAPE values in most iRT prediction tasks. These results imply that physicochemical features provide a significant advantage for iRT prediction and that incorporating them into the model architecture does not degrade MS2 prediction performance. The DIA data analysis showed that the Fusion w.o. scale model identified the highest number of proteins using peptides from Pan-Human library. This result suggests that atom count and global features may aid in protein group identification, particularly when regularized with an L1-norm. In contrast, the Fusion model identified the fewest peptides and proteins, while the Fusion w.o. L1 model yielded identified the most peptides and proteins using peptides from the combined Pan-Human + HeLa library. These findings indicate that incorporating physicochemical features is beneficial for DIA data analysis. However, applying L1-norm regularization, especially to amino acid scale features, may negatively affect performance. While the proposed method incorporates various physicochemical features of peptides, many other factors that may influence tandem MS measurement were not considered in this study. For example, difference in MS instruments and experimental settings, such as collision energy and collision cell type, can lead to variations in MS2 spectra 15 . By incorporating features related to these experimental conditions, the proposed deep learning model could potentially improve its predictive performance across diverse datasets generated under varying conditions. In future work, we plan tov extend the proposed method to account for different instrument types, experimental settings, peptide modifications, and diverse cell lines and species. Competing Interests N.L. and H.Yang are founding members and receive hold equity in Bionsight, Inc. with prior approval from Kangwon National University. All other authors declare to have no competing interests. Supporting Information Available supporting-information.pdf: deep learning model architectures Acknowledgement This work was supported by National Research Foundation of Korea (NRF) grants funded by the Korea government (MSIT) (Nos. RS-2021-NR061647, RS-2024-00358572, RS-2024-00336424). References (1). ↵ Aebersold , R. ; Mann , M. Mass spectrometry-based proteomics. Nature 2003 , 422 , 198 – 207 . OpenUrl CrossRef PubMed Web of Science (2). ↵ Domon , B. ; Aebersold , R. Mass spectrometry and protein analysis. Science 2006 , 312 , 212 – 217 . OpenUrl Abstract / FREE Full Text (3). ↵ Gillet , L. C. ; Navarro , P. ; Tate , S. ; Röst , H. ; Selevsek , N. ; Reiter , L. ; Bonner , R. ; Aebersold , R . Targeted data extraction of the MS/MS spectra generated by data-independent acquisition: A new concept for consistent and accurate proteome analysis . Molecular & Cellular Proteomics 2012 , 11 , O111 .016717. OpenUrl (4). ↵ Ting , Y. S. ; Egertson , J. D. ; Payne , S. H. ; Kim , S. ; MacLean , B. ; Käll , L. ; Aebersold , R. ; Smith , R. D. ; Noble , W. S. ; MacCoss , M. J . Peptide-centric proteome analysis: an alternative strategy for the analysis of tandem mass spectrometry data . Molecular & Cellular Proteomics 2015 , 14 , 2301 – 2307 . OpenUrl CrossRef PubMed (5). ↵ Schubert , O. T. ; Gillet , L. C. ; Collins , B. C. ; Navarro , P. ; Rosenberger , G. ; Wolski , W. E. ; Lam , H. ; Amodei , D. ; Mallick , P. ; MacLean , B. ; Aebersold , R . Building high-quality assay libraries for targeted analysis of SWATH MS data . Nature Protocols 2015 , 10 , 426 – 441 . OpenUrl CrossRef PubMed (6). ↵ Meyer , J. G . Deep learning neural network tools for proteomics . Cell Reports Methods 2021 , 1 , 100003 . (7). ↵ Moruz , L. ; Käll , L . Peptide retention time prediction . Mass Spectrometry Reviews 2017 , 36 , 615 – 623 . OpenUrl CrossRef PubMed (8). ↵ Petritis , K. ; Kangas , L. J. ; Ferguson , P. L. ; Anderson , G. A. ; Pasa-Tolić , L. ; Lipton , M. S. ; Auberry , K. J. ; Strittmatter , E. F. ; Shen , Y. ; Zhao , R. ; Smith , R. D . Use of artificial neural networks for the accurate prediction of peptide liquid chromatography elution times in proteome analyses . Analytical Chemistry 2003 , 75 , 1039 – 1048 . OpenUrl CrossRef PubMed (9). ↵ LeCun , Y. ; Bengio , Y. ; Hinton , G. Deep learning. Nature 2015 , 521 , 436 – 444 . OpenUrl CrossRef PubMed (10). ↵ Rumelhart , D. E. ; Hinton , G. E. ; Williams , R. J . Learning representations by backpropagating errors . Nature 1986 , 323 , 533 – 536 . OpenUrl CrossRef Web of Science (11). ↵ Hochreiter , S. ; Schmidhuber , J. Long short-term memory. Neural Computation 1997 , 9 , 1735 – 1780 . OpenUrl CrossRef PubMed Web of Science (12). ↵ Cho , K. ; van Merrienboer , B. ; Bahdanau , D. ; Bengio , Y . On the properties of neural machine translation: Encoder-decoder approaches . ArXiv preprint 2014 , arXiv:1409. 1259 . (13). ↵ Chung , J. ; Gulcehre , C. ; Cho , K. H. ; Bengio , Y . Empirical evaluation of gated recurrent neural networks on sequence modeling . ArXiv preprint 2014 , arXiv:1412. 3555 . (14). ↵ Schoenholz , S. S. ; Hackett , S. ; Deming , L. ; Melamud , E. ; Jaitly , N. ; McAllister , F. ; O’Brien , J. ; Dahl , G. ; Bennett , B. ; Dai , A. M. ; Koller , D . Peptide-spectra matching from weak supervision . ArXiv preprint 2018 , arXiv:1808.06576 . (15). ↵ Gessulat , S. ; Schmidt , T. ; Zolg , D. P . ; others Prosit: proteome-wide prediction of peptide tandem mass spectra by deep learning . Nature Methods 2019 , 16 , 509 – 518 . OpenUrl CrossRef PubMed (16). Guan , S. ; Moran , M. F. ; Ma , B . Prediction of LC-MS/MS properties of peptides from sequence by deep learning . Molecular & Cellular Proteomics 2019 , 18 , 2099 – 2107 . OpenUrl CrossRef PubMed (17). Tiwary , S. ; Levy , R. ; Gutenbrunner , P . ; others High-quality MS/MS spectrum prediction for data-dependent and data-independent acquisition data analysis . Nature Methods 2019 , 16 , 519 – 525 . OpenUrl CrossRef PubMed (18). Zeng , W.-F. ; Zhou , X.-X. ; Zhou , W.-J. ; Chi , H. ; Zhan , J. ; He , S.-M . MS/MS spectrum prediction for modified peptides using pDeep2 trained by transfer learning . Analytical Chemistry 2019 , 91 , 9724 – 9731 . OpenUrl CrossRef (19). ↵ Yang , Y. ; Liu , X. ; Shen , C. ; Lin , Y. ; Yang , P. ; Qiao , L . In silico spectral libraries by deep learning facilitate data-independent acquisition proteomics . Nature Communications 2020 , 11 , 146 . (20). ↵ Lin , Y. M. ; Chen , C. T. ; Chang , J. M . MS2CNN: predicting MS/MS spectrum based on protein sequence using deep convolutional neural networks . BMC Genomics 2019 , 20 , 906 . (21). ↵ Bouwmeester , R. ; Gabriels , R. ; Hulstaert , N . ; others DeepLC can predict retention times for peptides that carry as-yet unseen modifications . Nature Methods 2021 , 18 , 1363 – 1369 . OpenUrl CrossRef PubMed (22). ↵ Gasteiger , E. ; Hoogland , C. ; Gattiker , A. ; Duvaud , S. ; Wilkins , M. R. ; Appel , R. D. ; Bairoch , A. In The Proteomics Protocols Handbook ; Walker , J. M. , Ed.; Humana Press : Totowa, NJ , 2005 ; pp 571 – 607 . (23). ↵ Kyte , J. ; Doolittle , R. F . A simple method for displaying the hydropathic character of a protein . Journal of Molecular Biology 1982 , 157 , 105 – 132 . OpenUrl CrossRef PubMed Web of Science (24). ↵ Wilson , K. J. ; Honegger , A. ; Stötzel , R. P. ; Hughes , G. J . The behaviour of peptides on reverse-phase supports during high-pressure liquid chromatography . Biochemical Journal 1981 , 199 , 31 – 41 . OpenUrl Abstract / FREE Full Text (25). ↵ Vihinen , M. ; Torkkila , E. ; Riikonen , P . Accuracy of protein flexibility predictions . Proteins 1994 , 19 , 141 – 149 . OpenUrl CrossRef PubMed Web of Science (26). ↵ Emini , E. A. ; Hughes , J. V. ; Perlow , D. S. ; Boger , J . Induction of hepatitis A virus-neutralizing antibody by a virus-specific synthetic peptide . Journal of Virology 1985 , 55 , 836 – 839 . OpenUrl Abstract / FREE Full Text (27). ↵ Janin , J . Surface and inside volumes in globular proteins . Nature 1979 , 277 , 491 – 492 . OpenUrl CrossRef PubMed Web of Science (28). ↵ Cock , P. J. A. ; Antao , T. ; Chang , J. T. ; Chapman , B. A. ; Cox , C. J. ; Dalke , A. ; Friedberg , I. ; Hamelryck , T. ; Kauff , F. ; Wilczynski , B. ; de Hoon , M. J. L . Biopython: freely available Python tools for computational molecular biology and bioinformatics . Bioinformatics 2009 , 25 , 1422 – 1423 . OpenUrl CrossRef PubMed Web of Science (29). ↵ Guruprasad , K. ; Reddy , B. V. B. ; Pandit , M. W . Correlation between stability of a protein and its dipeptide composition: a novel approach for predicting in vivo stability of a protein from its primary sequence. Protein Engineering , Design and Selection 1990 , 4 , 155 – 161 . OpenUrl CrossRef (30). ↵ Bjellqvist , B. ; Basse , B. ; Olsen , E. ; Celis , J. E . Reference points for comparisons of two-dimensional maps of proteins from different human cell types defined in a pH scale where isoelectric points correlate with polypeptide compositions . Electrophoresis 1994 , 15 , 529 – 539 . OpenUrl CrossRef PubMed Web of Science (31). ↵ Goodfellow , I. ; Bengio , Y. ; Courville , A. Deep Learning ; MIT Press , 2016 ; http://www.deeplearningbook.org . (32). ↵ Wilhelm , M. et al. Deep learning boosts sensitivity of mass spectrometry-based immunopeptidomics . Nature Communications 2021 , 12 , 3346 . (33). ↵ Bruderer , R. ; Bernhardt , O. M. ; Gandhi , T. ; Xuan , Y. ; Sondermann , J. ; Schmidt , M. ; Gomez-Varela , D. ; Reiter , L . Optimization of experimental parameters in dataindependent mass spectrometry significantly increases depth and reproducibility of results . Molecular & Cellular Proteomics 2017 , 16 , 2296 – 2309 . OpenUrl CrossRef PubMed (34). ↵ Bruderer , R. ; Bernhardt , O. M. ; Gandhi , T. ; Miladinović , S. M. ; Cheng , L. Y. ; Messner , S. ; Ehrenberger , T. ; Zanotelli , V. ; Butscheid , Y. ; Escher , C. ; Vitek , O. ; Rinner , O. ; Reiter , L . Extending the limits of quantitative proteome profiling with dataindependent acquisition and application to acetaminophen-treated three-dimensional liver microtissues . Molecular & Cellular Proteomics 2015 , 14 , 1400 – 1410 . OpenUrl CrossRef PubMed (35). ↵ Cox , J. ; Mann , M . MaxQuant enables high peptide identification rates, individualized p . p.b.-range mass accuracies and proteome-wide protein quantification. Nature Biotechnology 2008 , 26 , 1367 – 1372 . OpenUrl CrossRef PubMed (36). ↵ Dau , T. ; Bartolomucci , G. ; Rappsilber , J . Proteomics using protease alternatives to trypsin benefits from sequential digestion with trypsin . Analytical Chemistry 2020 , 92 , 9523 – 9527 . OpenUrl CrossRef (37). ↵ Rosenberger , G. et al. A repository of assays to quantify 10,000 human proteins by SWATH-MS . Scientific Data 2014 , 1 , 140031 . View the discussion thread. Back to top Previous Next Posted June 12, 2025. Download PDF Email Thank you for your interest in spreading the word about bioRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following Regularized Deep Neural Networks for Combining Heterogeneous Features of Peptides in Data Independent Acquisition Mass Spectrometry Message Subject (Your Name) has forwarded a page to you from bioRxiv Message Body (Your Name) thought you would like to see this page from the bioRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share Regularized Deep Neural Networks for Combining Heterogeneous Features of Peptides in Data Independent Acquisition Mass Spectrometry Namgil Lee , Hojin Yoo , Dohyun Han , Heejung Yang bioRxiv 2025.06.09.658564; doi: https://doi.org/10.1101/2025.06.09.658564 Share This Article: Copy Citation Tools Regularized Deep Neural Networks for Combining Heterogeneous Features of Peptides in Data Independent Acquisition Mass Spectrometry Namgil Lee , Hojin Yoo , Dohyun Han , Heejung Yang bioRxiv 2025.06.09.658564; doi: https://doi.org/10.1101/2025.06.09.658564 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Bioinformatics Subject Areas All Articles Animal Behavior and Cognition (7617) Biochemistry (17633) Bioengineering (13856) Bioinformatics (41841) Biophysics (21399) Cancer Biology (18529) Cell Biology (25422) Clinical Trials (138) Developmental Biology (13352) Ecology (19860) Epidemiology (2067) Evolutionary Biology (24281) Genetics (15582) Genomics (22461) Immunology (17700) Microbiology (40295) Molecular Biology (17140) Neuroscience (88413) Paleontology (666) Pathology (2823) Pharmacology and Toxicology (4813) Physiology (7632) Plant Biology (15107) Scientific Communication and Education (2042) Synthetic Biology (4284) Systems Biology (9808) Zoology (2267)

Text is read by the "Ask this paper" AI Q&A widget below. Extraction quality varies by source — PMC NXML preserves structure cleanly, OA-HTML may include some navigation residue, and OA-PDF can have broken hyphenation. The publisher copy (via DOI) is the canonical version.

My notes (saved in your browser only)

Ask this paper AI returns verbatim quotes from the full text · source: preprint-html

Answers must be backed by verbatim quotes from this paper's full text. Hallucinated quotes are dropped automatically; if no verbatim passage answers the question, we say so. How this works

Citation neighborhood (no data yet)

We don't have any in-corpus citations linked to this paper yet. This is a recent paper (2025) — citers typically take a year or two to land, and the OpenAlex reference graph may still be filling in.

Source provenance

europepmc
last seen: 2026-05-20T01:45:00.602351+00:00
unpaywall
last seen: 2026-05-24T02:00:01.246996+00:00
License: CC-BY-4.0