Full text
88,566 characters
· extracted from
preprint-html
· click to expand
InstaNovo-P: A de novo peptide sequencing model for phosphoproteomics | bioRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-M677548'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search New Results InstaNovo-P: A de novo peptide sequencing model for phosphoproteomics Jesper Lauridsen , View ORCID Profile Pathmanaban Ramasamy , View ORCID Profile Rachel Catzel , View ORCID Profile Vahap Canbay , View ORCID Profile Amandla Mabona , View ORCID Profile Kevin Eloff , Paul Fullwood , Jennifer Ferguson , Annekatrine Kirketerp-Møller , Ida Sofie Goldschmidt , View ORCID Profile Tine Claeys , Sam van Puyenbroeck , View ORCID Profile Nicolas Lopez Carranza , View ORCID Profile Erwin M. Schoof , View ORCID Profile Lennart Martens , View ORCID Profile Jeroen Van Goey , View ORCID Profile Chiara Francavilla , View ORCID Profile Timothy Patrick Jenkins , View ORCID Profile Konstantinos Kalogeropoulos doi: https://doi.org/10.1101/2025.05.14.654049 Jesper Lauridsen 1 Department of Biotechnology and Biomedicine, Technical University of Denmark , Kongens Lyngby, Denmark 2 Department of Applied Mathematics and Computer Science, Technical University of Denmark , Kongens Lyngby, Denmark Find this author on Google Scholar Find this author on PubMed Search for this author on this site Pathmanaban Ramasamy 3 CompOmics, VIB Center for Medical Biotechnology, VIB , Ghent, Belgium 4 Department of Biomolecular Medicine, Faculty of Medicine and Health Sciences, Ghent University , Ghent, Belgium 5 Interuniversity Institute of Bioinformatics in Brussels, ULB-VUB , Brussels, Belgium 6 Structural Biology Brussels, Vrije Universiteit Brussel , Brussels, Belgium Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Pathmanaban Ramasamy Rachel Catzel 7 InstaDeep Ltd , 5 Merchant Square, London, W2 1AY, UK Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Rachel Catzel Vahap Canbay 1 Department of Biotechnology and Biomedicine, Technical University of Denmark , Kongens Lyngby, Denmark Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Vahap Canbay Amandla Mabona 7 InstaDeep Ltd , 5 Merchant Square, London, W2 1AY, UK Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Amandla Mabona Kevin Eloff 7 InstaDeep Ltd , 5 Merchant Square, London, W2 1AY, UK Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Kevin Eloff Paul Fullwood 8 Division of Molecular and Cellular Function, School of Biological Science, Faculty of Biology Medicine and Health (FBMH), The University of Manchester , M139PT Manchester, UK Find this author on Google Scholar Find this author on PubMed Search for this author on this site Jennifer Ferguson 8 Division of Molecular and Cellular Function, School of Biological Science, Faculty of Biology Medicine and Health (FBMH), The University of Manchester , M139PT Manchester, UK Find this author on Google Scholar Find this author on PubMed Search for this author on this site Annekatrine Kirketerp-Møller 1 Department of Biotechnology and Biomedicine, Technical University of Denmark , Kongens Lyngby, Denmark Find this author on Google Scholar Find this author on PubMed Search for this author on this site Ida Sofie Goldschmidt 1 Department of Biotechnology and Biomedicine, Technical University of Denmark , Kongens Lyngby, Denmark Find this author on Google Scholar Find this author on PubMed Search for this author on this site Tine Claeys 3 CompOmics, VIB Center for Medical Biotechnology, VIB , Ghent, Belgium 4 Department of Biomolecular Medicine, Faculty of Medicine and Health Sciences, Ghent University , Ghent, Belgium Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Tine Claeys Sam van Puyenbroeck 3 CompOmics, VIB Center for Medical Biotechnology, VIB , Ghent, Belgium 4 Department of Biomolecular Medicine, Faculty of Medicine and Health Sciences, Ghent University , Ghent, Belgium Find this author on Google Scholar Find this author on PubMed Search for this author on this site Nicolas Lopez Carranza 7 InstaDeep Ltd , 5 Merchant Square, London, W2 1AY, UK Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Nicolas Lopez Carranza Erwin M. Schoof 1 Department of Biotechnology and Biomedicine, Technical University of Denmark , Kongens Lyngby, Denmark Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Erwin M. Schoof Lennart Martens 3 CompOmics, VIB Center for Medical Biotechnology, VIB , Ghent, Belgium 4 Department of Biomolecular Medicine, Faculty of Medicine and Health Sciences, Ghent University , Ghent, Belgium Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Lennart Martens Jeroen Van Goey 7 InstaDeep Ltd , 5 Merchant Square, London, W2 1AY, UK Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Jeroen Van Goey Chiara Francavilla 1 Department of Biotechnology and Biomedicine, Technical University of Denmark , Kongens Lyngby, Denmark 8 Division of Molecular and Cellular Function, School of Biological Science, Faculty of Biology Medicine and Health (FBMH), The University of Manchester , M139PT Manchester, UK Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Chiara Francavilla Timothy Patrick Jenkins 1 Department of Biotechnology and Biomedicine, Technical University of Denmark , Kongens Lyngby, Denmark Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Timothy Patrick Jenkins For correspondence: tpaje{at}dtu.dk konka{at}dtu.dk Konstantinos Kalogeropoulos 1 Department of Biotechnology and Biomedicine, Technical University of Denmark , Kongens Lyngby, Denmark 9 Department of Bionanoscience, Delft University of Technology , 2629 HZ Delft, Netherlands 10 Kavli Institute of Nanoscience , 2629 HZ Delft, Netherlands Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Konstantinos Kalogeropoulos For correspondence: tpaje{at}dtu.dk konka{at}dtu.dk Abstract Full Text Info/History Metrics Data/Code Preview PDF ABSTRACT Phosphorylation, a crucial post-translational modification (PTM), plays a central role in cellular signaling and disease mechanisms. Mass spectrometry-based phosphoproteomics is widely used for system-wide characterization of phosphorylation events. However, traditional methods struggle with accurate phosphorylated site localization, complex search spaces, and detecting sequences outside the reference database. Advances in de novo peptide sequencing offer opportunities to address these limitations, but have yet to become integrated and adapted for phosphoproteomics datasets. Here, we present InstaNovo-P, a phosphorylation specific version of our transformer-based InstaNovo model, fine-tuned on extensive phosphoproteomics datasets. InstaNovo-P significantly surpasses existing methods in phosphorylated peptide detection and phosphorylated site localization accuracy across multiple datasets, including complex experimental scenarios. Our model robustly identifies peptides with single and multiple phosphorylated sites, effectively localizing phosphorylation events on serine, threonine, and tyrosine residues. We experimentally validate our model predictions by studying FGFR2 signaling, further demonstrating that InstaNovo-P uncovers phosphorylated sites previously missed by traditional database searches. These predictions align with critical biological processes, confirming the model’s capacity to yield valuable biological insights. InstaNovo-P adds value to phosphoproteomics experiments by effectively identifying biologically relevant phosphorylation events without prior information, providing a powerful analytical tool for the dissection of signaling pathways. Introduction Phosphorylation is a widespread protein post-translational modification (PTM) and the most abundant PTM in eukaryotes 1 – 3 . It plays crucial roles in enzyme activation, signal transduction, and transcription regulation, with aberrant activity linked to various pathological states, including cancer 4 – 6 . Consequently, detection of phosphorylation events is imperative for dissecting cellular pathways and describing disease states. Kinases, the enzymes responsible for phosphorylation, are significant drug targets and have found considerable success in therapeutic intervention 7 , 8 . Mass spectrometry has become the de facto approach for studying protein phosphorylation across different cell types and tissues under various conditions 9 – 13 . Phosphoproteomics, the field dedicated to studying phosphorylated proteins, has seen major advancements in sample preparation, instrumentation, and data analysis for phosphorylated peptide detection over the past decades 14 – 20 . The traditional approach for identifying phosphorylated peptides involves database searches based on the mass shift induced by phosphorylation in the sequence search space. However, this method has inherent limitations in modification localization ambiguity, and significantly expands the search space, search time and costs 21 – 23 . Additionally, target-decoy methods cannot detect non-canonical protein sequences or protein isoforms that are not included in the database search, restricting analysis to only known reference proteomes. Recent advances in de novo peptide sequencing using machine learning approaches have shown impressive performance in peptide detection, including peptides with modifications such as N-terminal acetylation and methionine oxidation 24 – 26 . These methods have the potential to address the shortcomings of database searches, provide complementary information, and be integrated into hybrid approaches to maximize insights into phosphorylation in investigated proteomes. Despite the abundance of publicly available PTM proteomics datasets, de novo peptide sequencing approaches have not been extensively adapted for PTM identification. Previous transformer-based methods for de novo peptide sequencing, including InstaNovo 24 , predominantly utilize auto-regressive architectures, predicting amino acids sequentially. Recently, Xia et al. developed AdaNovo 27 , incorporating conditional mutual information between spectra and peptides, successfully identifying amino acids carrying PTMs on benchmark datasets, however without including phosphorylation. PrimeNovo 28 , a model using non-autoregressive prediction, is currently the best model that supports detection of phosphorylated peptides. Here, we introduce the first application of a de novo peptide sequencing model that is trained on an extensive number of phosphoproteomic datasets and is exclusively adapted for phosphorylated peptide detection. We reprocess numerous phosphoproteomics experiments available in the PRIDE database 29 to make a fine-tuned, phosphorylation-specific version of InstaNovo, achieving substantial improvements in phosphorylated peptide identification accuracy. We validate the performance of our fine-tuned model across multiple datasets, and confirm its predictions with targeted proteomics. Lastly, we provide for the first time experimental validation of phoshorylated peptide predictions that went undetected by conventional database searches. Results InstaNovo-P is a phosphorylation-specific de novo sequencing model We utilized our previously developed de novo sequencing model, InstaNovo 24 , and adapted it to the task of phosphorylated peptide detection. InstaNovo is an autoregressive transformer model with an encoder-decoder architecture, trained on the ProteomeTools dataset 30 . During inference, InstaNovo interprets an MS/MS spectrum by translating it to the peptide sequence that produced the spectrum. To fine-tune InstaNovo for phosphorylated peptide prediction (i.e., to further train and adapt our model for the specific task of phosphorylated peptide sequencing), we curated a set of 29 reprocessed phosphoproteomics projects from Scop3P 31 (see Supplementary Table 1 for complete list of projects). After filtering for high localization probability ( > 0.80), our training dataset consisted of 2,760,939 PSMs mapping to 74,686 unique peptides. Three projects made up more than half of the PSMs with PXD000612 alone containing 837,540 or 31% of the PSMs ( Figure 1A ). When accounting for unique peptides per project, the analysed dataset was also dominated by a few large projects, although the distribution is slightly more even ( Figure 1B ). The distribution of phosphorylated amino acids is dominated by serine (S), making up 84.5%, while threonine (T) and tyrosine (Y) make up 12.0% and 3.4%, respectively ( Figure 1C ). To minimize data leakage, the dataset was split using GraphPart 32 into a train-valid-test split of 70-10-20%. Download figure Open in new tab Figure 1. Overview of training data and InstaNovo-P model. A) PSMs per PRIDE project reprocessed and used to train the data. B) Unique phosphorylated peptide sequences used to train the model per project. C) Percent of phosphorylated residue types in the combined train-valid-test set. D) InstaNovo-P architecture. The sequence embedding and linear prediction layer in the decoder head have been adapted to accommodate the prediction of log probabilities for phosphorylated tyrosine, serine and threonine tokens, in addition to the standard vocabulary and modifications of the base InstaNovo model. E) Training loss, validation loss, and peptide recall during fine-tuning with gradual unfreezing. The blue region had only the head and embedding layers unfrozen, the middle greed region was gradually unfrozen in the encoder layers, the final red region was gradually unfrozen in the decoder layers. We used the InstaNovo model architecture with two modifications. The sequence embedding layer and the linear head of the model had to be extended to accommodate three extra tokens for the possible phosphorylated residues of serine, threonine, and tyrosine ( Figure 1D ). To fine-tune, we utilized gradual unfreezing 33 of the model layers starting with the head and sequence embedding for 1 epoch. Then, instead of unfreezing the decoder, we gradually unfroze the encoder layers. Finally, the decoder layers were also gradually unfrozen, reducing the learning rate with increasing number of training steps ( Supplementary Figure S1A ). Interestingly, the model reached good validation performance after only fine-tuning the head, sequence embedding and encoder, but the decoder is also fine-tuned for the last performance gain ( Figure 1E , see Methods for further details). To assess the model performance on phosphorylated peptides, we used 3 evaluation datasets: the test set made from the split of the curated projects, here referred to as “Test”, a subset of project PXD009449 34 , here referred to as “21PTM”, and a dataset from an in-house experiment using T47D breast cancer cell line expressing or not Fibroblast Growth Factor Receptor 2 (FGFR2) upon growth factor treatment, here referred to as “FGFR2”. The three evaluation datasets contained different distributions of non-phosphorylated and phosphorylated peptides as well as different residue types ( Supplementary Figure S1B and S1C ). In addition to the evaluation on phosphorylated datasets, we also evaluate the model on 2 datasets of unmodified peptides: The original test set used by the base model, here referred to as “AC-PT”, and the unmodified version of 21PTM, here referred to as “21PTM-unmod”. In summary, we extended the InstaNovo transformer by adding three phosphorylation-specific tokens and fine-tuned on a 2.76 million PSM phosphoproteomics corpus, achieving strong validation performance using an encoder-first gradual unfreezing schedule. InstaNovo-P assigns phosphorylation events with high accuracy and precision We compared the performance of our model with the base InstaNovo model (version 0.1.4) on unmodified peptides from the AC-PT dataset (ProteomeTools Parts I-III) 24 and the 21PTM dataset 34 . InstaNovo-P increases peptide recall from 64.0% to 68.1% and from 70.0% to 71.4% for the AC-PT test set and 21PTM dataset (taking into account only unmodified peptides), respectively, compared to the base InstaNovo model ( Figure 2A ). This surprising observation indicates that the fine-tuning effort did not constrain the model to the detection of phosphorylated peptides only, but also improved the general performance. In other words, the model did not experience catastrophic forgetting even without mixing training data from target and source domain (i.e., unmodified vs modified peptides). Download figure Open in new tab Figure 2. Benchmarking and evaluation of InstaNovo-P predictions. A) Comparison of Instanovo and InstaNovo-P peptide recall on the AC-PT (Proteome Tools I-III) and 21PTM dataset. B) Comparison of PrimeNovo and InstaNovo-P peptide recall on the 21PTM dataset. C) Peptide recall for each test dataset for all peptides, and peptides with different number of phosphorylated sites per peptide. D) Peptide recall for each test dataset split by type of phosphorylated residue. E) Peptide recall in each dataset for different FDR thresholds as calculated by database grounding. F) Peptide recall for different FDR thresholds, split by phosphorylated residue type in the FGFR2 dataset. We then evaluated the performance on phosphorylated peptides of the 21PTM dataset. InstaNovo-P achieved a peptide recall of 81.4% ( Supplementary Table 2 ). In comparison, PrimeNovo 28 reported a peptide recall of 68% ( Figure 2B ). This increase of 13.4% compared to PrimeNovo indicates an excellent generalization to unseen phosphorylated peptides. Moreover, PrimeNovo was fine-tuned specifically on 21PTM, whereas InstaNovo-P evaluated 21PTM entirely as a holdout dataset. Next, we evaluated the performance of our model in relation to the number of phosphorylated sites present on phosphorylated peptides. Our evaluation datasets contain unmodified peptides, as well as peptides with 1, 2, or more phosphorylation sites ( Supplementary Figure S2A ). As expected, we see decreasing peptide recall with the increased number of phosphorylated sites in each peptide ( Figure 2C ). Unsurprisingly, we also observed a correlation of peptide recall with the composition of our training dataset. The highest recall was seen in serine phosphorylation events ( Supplementary Table 3 ), which were also the most abundant events in our training dataset ( Figure 2D ). The highest recall was seen for the 21PTM dataset, which contained only phosphorylated tyrosines. We speculate that the reason is due to high intensity signal, the nature of the synthetic peptides, and lower sample complexity. Limiting and estimating the number of false positives in proteomics searches is imperative for utility in biological studies and downstream validations. We therefore evaluated the model confidence threshold needed for different values of false discovery rate (FDR, Supplementary Figure S2B ). The confidence range distribution of predictions for all evaluation datasets was similar to the base model, and indicates that the model can inherently distinguish between correct and incorrect assignments ( Supplementary Figure S3 ). We observed highly diminishing peptide recall at 1% FDR ( Figure 2E ). Thus the best threshold that balanced recall and by extension identification performance with a satisfactory FDR was 5%. We also observed the same trend based on residue phosphorylation type. Peptide recall on the FGFR2 dataset by type of phosphorylated residue at 5% FDR was 44.4%, 30.2% and 22.6% for S, T and Y, respectively ( Figure 2F ). Altogether, these data indicate that InstaNovo-P generalizes well to phosphorylated peptides without catastrophic forgetting, while the predicted confidence scores effectively discriminate correct assignments, with a 5% FDR threshold offering the optimal trade-off between recall and reliability. InstaNovo-P performance is dependent on peptide properties We next evaluated the precision and recall of peptide predictions on the three different test datasets. As previously observed for our base InstaNovo model, we saw highly variable performance depending on the dataset evaluated ( Figure 3A and B ). InstaNovo-P achieved the best performance on 21PTM, which was composed of synthetic peptides. We recorded lower recall in our test and FGFR2 dataset, which originated from other projects with endogenous peptides. We noted that performance was dependent on to peptide length ( Figure 3C and D ). InstaNovo-P exhibited lower peptide recall with increasing peptide length, which affected the dataset total recall. Even though the overall recall for 21PTM was higher than that of the test set and FGFR2 dataset, we found that the recall for shorter peptides (under 15 amino acids) was almost identical across datasets ( Figure 3D ). The FGFR2 dataset included both a phospho-enriched and an unmodified version of several peptides. In addition to those two, we also took the phospho-enriched FGFR2 and “de-modified” it (changing the phosphorylated sites to their non-modified version). A prediction would usually contain a phosphorylated site, but by de-modifying it, we could infer the added difficulty of phosphorylation localization alone. We found that InstaNovo-P performed well across all three types, however phosphorylated peptide identification was more difficult than for unmodified peptides ( Figure 3E ). For longer peptides, the recall performance was noticeably higher for the FGFR2 dataset when compared to the two other evaluation datasets, with recall also showing dependence on the number of phosphorylated sites for each peptide ( Figure 3F ) . Download figure Open in new tab Figure 3. Prediction accuracy is related to peptide properties. A) Precision-recall curves for each test dataset. Horizontal line indicates 5% FDR threshold as calculated by database grounding. B) Recall-coverage curve for each of the test datasets. C) Composition of test datasets as assessed by peptide length. D) Recall per bin for each test dataset. E) Precision recall curve for the FGFR2 dataset including Demod, which has all modifications in both predictions and targets removed, and Unmod, which only contains unmodified peptides. Horizontal line indicates 5% FDR threshold. F) Peptide recall for each test dataset, with separate curves for peptides that have one and two or more phosphorylated sites. In sum, InstaNovo-P performance varied markedly by dataset and peptide characteristics, while recall also trended downward as the number of phosphorylation sites per peptide increased. Our results suggest that our recall differences between the three evaluated datasets were predominantly due to differences in peptide attributes. InstaNovo-P is a robust tool for proteome-wide identification of phosphorylation events We used our FGFR2 dataset to map our model’s predictions to the proteome, providing another validation for our predictions, given the very low likelihood of an incorrect prediction mapping to the proteome. We also compared our results to database search results using MaxQuant. We assessed our false discovery rate using the database search identifications in the labeled space (scans in common with database results) or the full search space (all scans in the dataset). We noticed that InstaNovo-P predicted substantially more phosphorylated peptides at 5% FDR when evaluated in the full search space rather than the labeled space alone, and that the gains are similar across different perturbations and time points in our experiments. Overall, InstaNovo-P detected 97.1% more peptides ( Figure 4A ). On average, InstaNovo-P detected 3, 068( ± 495) peptides, which was expanded to 5, 436( ± 770) peptides when expanding our search to the full search space, i.e., also scans that the database did not identify ( Figure 4B ). We mapped these peptides to 21.6% more proteins in the proteome compared to the database search in the same dataset (4B). The identified peptides mapped to 1, 331( ± 209) proteins on average in the labeled space, which increased to 1, 544( ± 239) proteins in the full search space ( Figure 4D ). Download figure Open in new tab Figure 4. InstaNovo-P expands identifications and localization certainty. A) Venn diagram of the number of phosphorylated peptides in the database search space (labeled), and predicted phosphorylated peptides from InstaNovo-P in full search space of the FGFR2 dataset. B) Proteins detected with database search and InstaNovo-P at an estimated 5% FDR threshold in labeled and full search space. C) Peptides in database search and InstaNovo-P with 5% FDR for each condition in the experiment. D) Proteins detected with database search and InstaNovo-P with 5% FDR for each condition in the experiment. E) Localization precision - peptide recall curve for different localization certainty thresholds according to database search. Horizontal line indicates 5% FDR as calculated by database grounding. F) Normalized density of true positive (True) and false positive (False) predictions, as ranked by localization certainty according to the database search. G) Mapping of phosphorylated sites detected with InstaNovo-P (top) and database search (bottom) in protein TP53 for all experimental conditions. Phosphorylated sites that are unique to either method are extended. The localisation accuracy of InstaNovo-P, as assessed by recall, indicated that the model is adept at recognizing and predicting phosphorylation events for peptides containing multiple potential phosphorylation sites. The recall observed, as well as the recall at an estimated 5% FDR, positively correlated with database localization certainty ( Figure 4E ). However, peptides with low localization certainty included in the database search results might contain incorrect localization certainty that negatively affects peptide recall. This is substantiated by the trend of database localization probability as a function of correct and incorrect assignments by InstaNovo-P in the same dataset ( Figure 4F ). We then turned our attention to a number of specific phosphorylation sites in proteins that are critical components of the cell cycle. We first investigated phosphorylated peptide predictions for sites in the protein TP53, an important cell cycle checkpoint 35 . InstaNovo-P predicted 35 phosphorylated sites in total, of which 4 contain phosphorylation sites in TP53 that went undetected in the database search space ( Figure 4G ). InstaNovo-P predicted 31 phosphorylation sites in common with the database search, while the search predicted 8 sites that InstaNovo-P did not detect. We also evaluated predicted peptides and phosphorylation sites not detected by database search in lamin A/C (LMNA), a protein that is known to be phosphorylated during mitosis and is thought to be involved in chromatin stability and regulation 36 . In the full search space, InstaNovo-P predicted 6 more peptides containing 2 more unique phosphorylated sites that are not present in the labeled space ( Supplementary Figure S4 ). Collectively, proteome-wide mapping in the FGFR2 dataset shows that InstaNovo-P not only nearly doubles identified phosphorylated peptides at 5% FDR by extending our search to the full search space, but also retains high localization accuracy that scales with database search localization certainty. Importantly, it uncovers novel phosphorylation sites, demonstrating potential in revealing additional phosphorylation sites beyond database searches. InstaNovo-P predicts phosphorylation sites that contribute meaningful biological insights To evaluate potential novel insights provided by our model we used a subset of the FGFR2 dataset and compared T47D cells, either wild type (WT) or CRISPR/CAS9-depleted of FGFR2 (KO), upon treatment with the specific FGFR2 ligand FGF7 (F7) 37 for different timepoints (40 min or 72 hours). Each of these conditions contained only phospho-enriched samples, eluted in 2 separate fractions per column. We mapped InstaNovo-P predictions from these experimental files in the human proteome, and assessed their presence in the labeled space. InstaNovo-P identified 86.75% more peptides that contained phosphorylated sites compared to the labeled space ( Figure 5A ). The distribution of these peptides matched well with the standard peptide distribution of human peptides in MS data 24 and the database search space ( Figure 5B ). We also explored InstaNovo-P predictions in the labeled and full search space in relation to the database search. At 5% FDR, we predicted more than double the peptides compared to the model predictions that overlap with the database search in the same threshold ( Figure 5C ). Download figure Open in new tab Figure 5. InstaNovo-P predicts biologically relevant phosphorylation events. A) Venn diagram of the number of phosphorylated peptides in the database search space (labeled), and predicted phosphorylated peptides from InstaNovo-P in the FGFR2 dataset, F7 treated subset at 5% FDR. B) Histogram of unique phosphorylated peptide predictions at 5% FDR as a function of peptide length. C) Peptides predicted by InstaNovo-P 5% FDR, labeled by presence or absence in database search. D) Unique peptides predicted by InstaNovo-P at 5% FDR in WT and KO replicates, when treated with F7 for 40 minutes. E) Selected enriched gene ontology terms in the two conditions (F7 treated WT vs KO, and WT treated vs untreated, at 40 minutes) of the FGFR2 dataset. F) Best scan for the phosphorylated peptide “EIQNGNLHEpSDSESVPR” in the WT cells. G) Fragment ion transitions for the same precursor peptide for endogenous (light peptide, left) and reference peptide (heavy peptide, right) When comparing the predictions between F7-treated KO and WT cells at the 40 min, InstaNovo-P identified 695 peptides that were unique for WT ( Figure 5D ) at 5% FDR. These peptides may represent the core of F7-FGFR2 signaling in our experimental model as they were found in the WT but not in the KO cells. Out of these, 153 were not found in the database search space, with 138 of them found only in this condition and timepoint. Importantly, 10 out of 10 of these novel top predicted phosphorylated sites, when ranked by number of PSMs, represented known phosphorylated sites in PhosphoSitePlus 10 . As an example, we identified a phosphorylation site at S402, in the C-terminal peptide of SAC31, a protein predicted to be involved in centrosome duplication and mitotic progression (the most commonly phosphorylated residue in this protein according to experiments integrated in PhosphoSitePlus). This suggests that our model is able to identify known and validated phosphorylated peptides, providing complementary information and substantially improving phosphoproteomics searches. The comparison between F7-treated vs untreated WT cells revealed 2,340 phosphorylated peptides unique for the treated samples, with 417 identified only by InstaNovo-P, and 368 of these identified only in that condition. Similarly to the WT versus KO comparison, all 10 top ranked novel peptides contained phosphorylated sites present in PhosphoSitePlus. Three of these 10 peptides mapped to nucleophosmin (NPM1), and contain the same phosphorylated site at position S125, possibly phosphorylated by the kinases Aurora A or B. This site is the most frequent phosphorylation event in NPM1, and mutations in this position show mitotic defects 38 . This phosphorylation event causes the protein to shift to a monomeric state instead of a pentameric 39 , and re-localize from the nucleolus to midbody 38 . However, as none of these peptides identified on NPM1 are tryptic, they would not have been identified with the standard database search settings. These peptides differed by one residue at the N-terminus (starting with GSGPVHISGQHIVAVEEDAES(+79.97)EDEEEEDVK), with several cleavage events downstream of this peptide, hinting at aminopeptidase degradation of the protein in a sequential way. Quantification of these peptides corroborated this observation, with the truncated peptidoforms found in increasing or diminishing abundance ( Supplementary Figure S5 ). Although caspases -3, -7 and -8 can cleave NPM1 during apoptosis 40 , no cleavage or aminopeptidase activity has been reported for that protein region. We also observed an unmodified, cleaved product in the acidic region of NPM1 (161-188) with cleavage at position 180. This suggests that InstaNovo-P can detect modified, non-tryptic peptides and provide another layer of information in proteomics searches. We further scrutinized the novel predictions in these two experimental conditions, by performing gene enrichment analysis for the proteins that the novel phosphorylated peptides map to with Metascape 41 and default settings with human background. This analysis revealed expected processes significantly enriched in these conditions such as regulation of the cell cycle and RHO GTPase activity ( Figure 5E ). Overall, these results indicate that our model successfully predicted phosphorylated peptides that recapitulate the known biological processes downstream of F7-FGFR2 signaling in T47D cells 42 – 44 . To confirm our observations, we validated the predicted phosphorylated serine residue on kinectin-1 (KTN1) at position 75 with targeted proteomics in newly collected samples for the same conditions. We designed a parallel reaction monitoring (PRM) assay for the phosphorylated peptide “EIQNGNLHEpSDSESVPR” within residue 66 to 82 we identified with our predictions but not with database search. We detected the peptide precursor with the 98 dalton mass shift on the y8 ion which indicated the loss of the phosphoric acid group through fragmentation, confirming our de novo predictions ( Figure 5F ). Furthermore, we could show that the ion chromatograms of the reference peptide (isotopically labeled with C13 arginine at the C-terminus) and the retention time aligned well with the endogenous peptide within WT T47D breast cancer cells ( Figure 5G ). These results show that applying InstaNovo-P to FGFR2 phosphoproteomics uncovered hundreds of additional phosphorylation sites, many of which map to known signaling proteins and are catalogued in phosphorylation databases. Enrichment analysis tied these discoveries to cell-cycle events, and targeted proteomics confirmed the potential of InstaNovo-P in predicting biologically meaningful phosphorylation events that traditional database searches miss. Discussion In this work, we introduce and evaluate the phosphorylation-specific de novo peptide sequencing model, InstaNovo-P. We find InstaNovo-P to increase performance of the base model, InstaNovo, in unmodified peptides in the AC-PT test set and 21PTM-unmod dataset. This indicates that catastrophic forgetting during fine-tuning was mitigated as the model not only maintains, but improves performance in the source domain. We believe that the encoder-first gradual unfreezing schedule contributed to this effect. By unfreezing the encoder layers before the decoder, we allowed the model to adapt to phosphorylation-specific features while preserving generalisable representations from the base model. This strategy may have helped avoid catastrophic forgetting, as reflected in improved recall on both phosphorylated and unmodified peptides. Further, our model was not only successful at improving detection of phosphorylated peptides, but also improving general sequencing performance. We speculate that this performance gain stems from InstaNovo not having achieved ceiling performance and benefitting from further training. It is also possible that seeing examples of modified peptides, or data from diverse origins, improves learning of unmodified peptide representations. InstaNovo-P delivered state-of-the-art recall on phosphorylated peptide PSMs. InstaNovo-P was able to detect novel peptides with high confidence, thus discovering phosphorylated peptides not detected by database search. Importantly, these phosphorylation events have been detected in previous proteomics experiments, validating our approach and findings. However, database search still identifies a substantial number of peptides not detected with InstaNovo-P, at least not as the top prediction. This is potentially due to lack of a complete fragment ion series, low fragment ion signal, chimeric spectra, or single residue errors in the peptide predictions. Overall, we believe that InstaNovo-P can be used in tandem with database searches to increase phosphorylated peptide identification. We also find a substantial number of InstaNovo-P predictions that map to proteins in the reference proteome while not present in database results, further validating our predictions. Additionally, its confidence scores correlate with true FDR and database localization certainty, offering a powerful complementary layer of site-localization information. We observed several instances of our model disagreeing with database localization assignment, and validating these cases will be part of future work. We believe that tighter integration with database search engines, where InstaNovo-P could rescue ambiguous or unassigned spectra, will maximize phosphoproteome coverage and accelerate biological discovery. InstaNovo-P shows diminished performance in peptides with more than one phosphorylation sites. Similarly, we see a correlation between peptide recall and peptide length. Longer sequences tend to result in increased sequence complexity and ambiguity, as well as potential missed cleavages, which make them harder to identify. Additionally, longer peptides often suffer from lower ionization efficiency, making them more difficult to detect in mass spectrometry. Unsurprisingly, we observe best performance on phosphorylation events in serine residues and worst performance on tyrosine phosphorylation events, reflecting our training data composition. This bias, where performance drops for peptides with more phosphorylation sites and less abundant phosphorylation events, can be explained by the increased complexity of the task and imbalance of data. We expect that our performance in threonine and tyrosine phosphorylation events will catch up once the model has been trained with adequate data for these residues. InstaNovo-P exhibits its highest recall on the 21PTM test set, potentially because it is synthetic peptides closely resembling our training dataset for the base model. Admittedly, it is challenging to compare models trained on different datasets with potential for data leakage. In this work, we have implemented a strict homology based split to control data leakage. InstaNovo-P has fixed inference, in contrast to database search were the space increases with addition of modifications. This might be beneficial for computational costs and search times, especially when large reference proteomics and additional modifications are considered. InstaNovo-P is only compatible with data dependent acquisition, as it relies on additional information about precursor mass and charge. Adapting the model, or preprocessing data independent acquisition spectra to be compatible with the model will be part of future directions. Similarly, we expect part of future work to be the interpretation of InstaNovo-P weights and attention, that can possibly inform on fragmentation events and ions that enable the model to accurately predict phosphorylated peptides. Finally, in this study we describe a reproducible framework providing a blueprint for model fine-tuning for different modifications or application domains with substantial biological utility, thus facilitating broader advancements within proteomics research. Methods Data preprocessing For a description of the All-Confidence ProteomeTools (AC-PT) dataset used to train the base model, see the InstaNovo 24 paper. The dataset used for fine tuning is comprised of a collection of reprocessed PRIDE projects in Scop3P 31 (For a list of the projects, see Supplementary Table 1 ). The dataset originally contains 4, 053, 346 PSMs. To only fine-tune on high confidence PSMs, the dataset is filtered at a confidence threshold of 0.80, reducing it to 2, 760, 939 PSMs, representing 74, 686 unique peptide sequences. Most of the data is of human origin, except for PXD005366 and PXD000218, which contain a mix of human and mouse. All PSMs that were used to train the model contained at least one phosphorylated site, while 169, 114 PSMs (6%) contained oxidated methionine. The entries were acquired using a mixture of instruments, whereas the AC-PT dataset was acquired using only Orbitrap Fusion. To partition the fine-tuning dataset into training, validation and test, GraphPart 32 , an algorithm for homology partitioning, was applied on the set of unique peptide sequences. GraphPart was set to use MMseqs2 45 with a partitioning threshold of 0.8 and a train-validation-test ratio of 0.7 / 0.1 / 0.2. Of the 74, 686 unique sequences, 390 were removed by GraphPart, reducing the total number of PSMs to 2, 691, 117 in a 2, 008, 923 / 232, 641 / 449, 553-split, although during training, a random subset of only 2% of the validation set was used in order to reduce computation. The 21PTM 34 dataset consists of around 5000 synthetic peptides carrying 21 different post-translational modifications as well as their unmodified counterparts. We used the phosphorylated and unmodified subsets. To minimize data leakage, we removed any sequences that overlapped with the fine-tuning training set. We removed sequences that had identical sequences, even if they had different phosphorylation sites. For the phosphorylation subset, we dropped 4094 sequences, going from 45252 to 41158. For the unmodified subset, we dropped 2188 sequences, going from 34823 to 32635. Even though we removed sequences that overlapped with our phosphorylation training-set, 21PTM still contains a significant overlap with the training set of AC-PT, used to train the base InstaNovo model. Specifically, 56.15% of unmodified sequences in 21PTM overlap with AC-PT, owing to the fact that 21PTM is part of the ProteomeTools project. Of course, none of the phosphorylated sequences overlap. For evaluation of predictions in all datasets, we remapped carbamidomethylated cysteines (C[+57.02]) to “C”, and isoleucine (I) to leucine (L). Fine-tuning Before fine-tuning, we reinitialized the head and sequence embedding of the model decoder, as they had to contain the extended vocabulary of the 3 phosphorylation tokens. All parameters in the specific layers were randomly initialized, therefore not carrying over any information from previous learning. To fine-tune the base InstaNovo 24 model we implemented a novel encoder-first gradual unfreezing fine tuning schedule. This gradual unfreezing 33 approach is typically implemented by first freezing the entire model, and then unfreezing layers, starting at the head, then progressing from the top of the decoder to the bottom of the decoder and sequence embedding, then from top to bottom through the encoder. We differ from this approach by first unfreezing the head and sequence embedding, then gradually the encoder, then gradually the decoder. We also explored other methods of fine-tuning, including full-model whereby the entire model is unfrozen from the start, head-only where only the final classifier layer is unfrozen and head-then-full where once the head has converged, the remainder of the model is unfrozen. When applying head-only fine-tuning, we also had to reinitialise the sequence embedding layer due to the different vocabulary sizes of the fine-tuning dataset and the dataset the base model was trained on. The alternative fine-tuning approach of encoder-first gradual unfreezing was motivated by the fact that the more standard fine-tuning approaches quickly led to overfitting. We find the Slanted Triangular Learning Rate (STLR) 33 scheduler to be most performant for this encoder-first gradual unfreezing. The learning rate is warmed up for the first 10% of the STLR period, and annealed for the remainder of the period (See Supplementary Figure S1A ). We applied the STLR scheduler in a cyclical manner, whereby each phase of unfreezing restarts the STLR schedule, thus allowing for all frozen parameters to warmup. This warmup strategy was found to be most performant (see Supplementary Table 4 for hyperparameters and Supplementary data for the learning-rate schedule used). The best model was found in the final epoch at step 273,725 at a validation loss of 0.3949. The primary Python packages used for fine-tuning were PyTorch 46 , Lightning 47 , and Fine-Tuning Scheduler 48 to manage the gradual unfreezing and learning rate scheduling. Metrics We evaluated our model using peptide-level metrics: peptide precision, peptide recall and Area-Under-Curve (AUC) of precision-recall curves and recall-coverage curves. We used exact string matching to decide whether two peptides were equal. For further details, see the InstaNovo 24 paper. Cell culture and FGFR2 depletion The human epithelial cell line T47D (HTB-133) was purchased from ATCC and authenticated shortly before the samples were collected through short tandem repeat (STR) analysis of 21 markers by Eurofins Genomics, yielding positive results. The cells were routinely monitored monthly for mycoplasma contamination using a PCR-based detection assay (Venor® GeM–Cambio). T47D cells were cultured in RPMI 1640 Medium, GlutaMAX™ Supplement (Gibco), supplemented with 2 mM L-glutamine, 100 U/ml penicillin, 100 µ g/ml streptomycin, and 10% fetal bovine serum. For gene editing, guide RNAs (crRNA) specific to FGFR2 (FGFR2 crRNA 1: GCCCTACCTCAAGGTTCTCA; FGFR2 crRNA 2: ACCTTGAGAACCTTGAGGTA) were synthesized (IDT) and combined with a common trans-activating crRNA (Alt-R® CRISPR-Cas9 tracrRNA, IDT) to form a functional ribonucleoprotein (RNP) duplex. The resulting complexes were pre-incubated with Cas9 nuclease (Alt-R® S.p. Cas9 Nuclease V3, IDT) and transiently transfected into parental T47D cells using Viromer® CRISPR transfection reagent (Cambridge Bioscience), following the manufacturer’s protocol. Colonies were subsequently selected and screened for genomic editing efficiency using the Inference of CRISPR Edits (ICE) analysis tool provided by Synthego. Successful knockout of FGFR2 protein expression was validated by Western blot analysis 19 . T47D were serum starved overnight and treated with 100 ng/mL of FGF7 or FGF10 from Peprotech for 40 min or 72h before lysis for MS analysis. Proteomics sample preparation T47D samples for proteomics and phosphoproteomics were prepared as follows 19 : Cells were washed with PBS and lysed in ice-cold 1% Triton lysis buffer supplemented with Pierce protease inhibitor tablets (Life Technologies) and phosphatase inhibitors (5 nM Na3VO4, 5 mM NaF, and 5 mM β -glycerophosphate). Proteins were precipitated overnight at ™ 20 ° C in a four-fold excess of ice-cold acetone. Acetone-precipitated proteins were then solubilized in a denaturation buffer consisting of 6 M urea and 2 M thiourea in 10 mM HEPES (pH 8). Cysteines were reduced with 1 mM dithiothreitol (DTT) and alkylated with 5.5 mM chloroacetamide (CAA). Proteins were subsequently digested using endoproteinase Lys-C (Wako, Osaka, Japan) followed by sequencing-grade modified trypsin (Sigma), and the digestion reaction was quenched using 1% trifluoroacetic acid (TFA). Peptides were purified using reversed-phase Sep-Pak C18 cartridges (Waters, Milford, MA) and eluted with 50% ACN. A small fraction (1%) of the eluted peptides was reserved for proteome analysis. These peptides were evaporated using a speed vacuum, resuspended in 40 µ L of 0.1% TFA, 5% ACN, and loaded onto C18 STAGE-tips. For phosphoproteomic analysis, 6 mL of 12% TFA in ACN was added to the remaining peptides, which were then enriched using TiO 2 beads (5 µ m, GL Sciences Inc., Tokyo, Japan). The beads were suspended in 20 mg/mL 2,5-dihydroxybenzoic acid (DHB), 80% ACN, and 6% TFA, and the samples were incubated in a 1:2 (w/w) sample-to-bead ratio for 15 min with rotation. After centrifugation for 5 min, the supernatant was collected and subjected to a second incubation with a two-fold dilution of the previous bead suspension. Elution of phosphorylated peptides was done with 20 µ L of 5% NH 3 , followed by 20 µ L of 10% NH 3 in 25% ACN, which were evaporated to a final volume of 5 µ L in a speed vacuum. The concentrated phosphorylated peptides were acidified by adding 20 µ L of 0.1% TFA and 5% ACN, and loaded onto C18 STAGE-tips. Peptides were eluted from STAGE-tips in 20 µ L of 40% ACN, followed by 10 µ L each of 60% ACN and 100% ACN, then reduced to 5 µ L by SpeedVac. Finally, 5 µ L of 0.1% FA and 5% ACN was added. A 1 µ L aliquot of the sample (for proteome analysis) or a 3 µ L aliquot (for phosphoproteome analysis) was transferred to a 5 µ L sample loop and loaded onto the column at a flow rate of 300 nL/min with 5% B for 5 and 13 minutes, respectively. After sample loading, the loop was removed from the flow path, and the flow rate was decreased from 300 nL/min to 200 nL/min within 1 minute, with the mobile phase adjusted to 7% B. Mass spectrometry data acquisition and analysis Purified peptides were analyzed by LC-MS/MS using an UltiMate® 3000 Rapid Separation LC system (RSLC, Dionex Corporation, Sunnyvale, CA) coupled to a Q Exactive HF mass spectrometer (Thermo Fisher Scientific, Waltham, MA). Mobile phase A consisted of 0.1% formic acid (FA) in water, and mobile phase B contained 0.1% FA in acetonitrile (ACN). Chromatographic separation was performed using a 75 mm × 250 µ m inner diameter, 1.7 mM CSH C18 analytical column (Waters). Peptides were separated using a linear gradient as follows: 7–18% B over 64 minutes, 18–27% B over 8 minutes, and 27–60% B in 1 minute. The column was subsequently washed at 60% B for 3 minutes and re-equilibrated for 6.5 minutes. At 85 minutes, the flow rate was increased back to 300 nL/min until the end of the run at 90 minutes. Mass spectrometry data were acquired over 90 minutes in positive ion mode using data-dependent acquisition. Peptides were automatically selected for fragmentation using a Top8 method (phosphoproteome) or Top12 method (proteome), based on precursors with charge states of 2, 3, or 4 and an m/z range of 300–1750 Th. Dynamic exclusion was set to 15 s. MS1 resolution was 120,000, with an AGC target of 3 × 10 6 and a maximum fill time of 20 ms. For MS2, resolution was set to 60,000 with an AGC target of 2 × 10 5 and a maximum fill time of 110 ms (Top12), or 30,000 with the same AGC target and a 45 ms fill time (Top8). The isolation window was set to 1.3 Th, and normalized collision energy was 28. Raw data were analyzed using the MaxQuant software suite ( https://www.maxquant.org ; version 1.6.2.6) with the integrated Andromeda search engine. Proteins were identified by searching the HCD-MS/MS peak lists against a target/decoy version of the human UniProt Knowledgebase database, including complete proteome sets and isoforms (version 2019; https://uniprot.org/proteomes/UP000005640_9606 ), supplemented with commonly observed contaminants such as porcine trypsin and bovine serum proteins. Tandem mass spectra were matched with a mass tolerance of 7 ppm for precursor ions and 0.02 Da (or 20 ppm) for fragment ions. Carbamidomethylation of cysteine was set as a fixed modification. Variable modifications included protein N-terminal acetylation, N-pyro-glutamine formation, oxidized methionine, and phosphorylation of serine, threonine, and tyrosine for phosphoproteome analyses. For proteome datasets, variable modifications included N-terminal acetylation, oxidized methionine, and deamidation of asparagine and glutamine. Site localization probabilities were computed by MaxQuant using the PTM scoring algorithm. Data were filtered by posterior error probability to ensure an FDR below 1% at the levels of peptides, proteins, and modification sites. Only peptides with an Andromeda score greater than 40 were included in the final analysis. Quantification of InstaNovo-P predictions Quantification was performed using AUC calculation for the predictions at their measured mass charge (m/z), with a tolerance of ±0.05. PSMs with identical sequences, but a difference in mass of more than ±0.05, were considered different PSMs. For identical PSMs, the retention time (RT) and m/z from the first identified PSM was used for quantification. Boundaries for the AUC were defined to limit cross-interference as the largest window between local minima within a ±1% RT window of the identification RT. Quantification was performed on MS1 spectra. Noise was calculated for each experiment, as the mean of the lowest 5% intensities from all MS2 spectra in the given file, and then subtracted from each intensity measurement during calculation of AUC. PSMs were grouped across experiment files if mass was within tolerance of ±0.05 and sequences and modifications were identical. Each group was then considered as one peptide. Peptide mass was determined by the average mass of the combined PSMs, and confidence was assigned as the highest confidence of the peptide’s PSMs. Targeted proteomics validation Confluent (80%) T47D breast cancer cells were starved in FBS-free medium for 24 hours. Subsequently cells were either treated with 100 ng/mL FGF7 or kept untreated for 40 minutes. Cells were placed on ice and 100 µL of lysis buffer (4 M GuHCl and 100 mM HEPES (pH 8.0)) were added together with Pierce protease inhibitor tablets (Life Technologies) and cells were scraped off. Scraped cells were transferred into a fresh tube, heated to 95 ° C for 5 minutes, and sonicated at 30% for 30 seconds while kept on ice. Lysed cells were centrifuged at 13,000 x g for 15 minutes at 4 ° C and the supernatant was transferred into a fresh tube. The Bradford assay was used to determine protein concentration and 5 µ g of protein lysate were used for reduction with Tris-(2-carboxyethyl)-phosphin (TCEP) (5 mM final concentration) and alkylation with CAA both carried out at 400 rpm, 65 ° C for 30 minutes. After reduction and alkylation, samples were diluted 8-fold to reduce the GuHCl concentration to 0.5 M using 100 mM HEPES (pH 8.0) and Lys-C (Wako, Osaka, Japan) (1:100) was added for 4 hours followed by overnight digestion with sequencing-grade modified trypsin (Sigma) at 37 ° C. The reaction was quenched using 1% TFA. Peptides were purified using Solaµ plates (Thermo Fisher Scientific, Waltham, MA). Purified peptides and phosphorylated peptides were analyzed by LC-MS/MS using an Vanquish LC system (Thermo Fisher Scientific, Waltham, MA) coupled to a Exploris 480 mass spectrometer (Thermo Fisher Scientific, Waltham, MA). Mobile phase A consisted of 0.1% formic acid (FA) in water, and mobile phase B contained 0.1% FA in 80% acetonitrile (ACN). Chromatographic separation was performed using a µPAC HPLC analytical column (Thermo Fisher Scientific, Waltham, MA). Peptides were separated using a gradient as follows: 1-10% B over 4.1 minutes, 10.1–22.5% B over 19 minutes, 22.5-45% over 8.4 minutes, and 45-95% over 6.5 minutes. The flow rate is kept at 750 nL/min over the initial 4.1 minutes and changed to 250 nL/min over 33.9 minutes. An initial data-dependent acquisition (DDA) strategy is used to acquire the spectral library of selected peptides and different charge states. The mass list for the reference heavy peptides is DINNIDYYK (583,284402 m/z), DINNIDYYK (389,192027 m/z), DINNIDYYK (292,145839 m/z), DINNIDY[+80]Y[+80]K (663,250733 m/z), DINNIDY[+80]Y[+80]K (442,502914 m/z), DINNIDY[+80]Y[+80]K (332,129005 m/z), EIQNGNLHESDSESVPR (960,949709 m/z), EIQNGNLHESDSESVPR (640,968898 m/z), EIQNGNL-HESDSESVPR (480,978493 m/z), EIQNGNLHES[+80]DSESVPR (1000,932875 m/z), EIQNGNLHES[+80]DSESVPR (667,624342 m/z), EIQNGNLHES[+80]DSESVPR (500,970075 m/z), GISSLPR (370,220455 m/z), GISSLPR (247,149395 m/z), GISSLPR (185,613865 m/z), GISS[+80]LPR (410,20362 m/z), GISS[+80]LPR (273,804839 m/z), GISS[+80]LPR (205,605448 m/z). The scans were unscheduled and a dependent scan was performed on the most intense ion if no target was found. For MS1, resolution was set to 120000 with an AGC target of 300% or 3×10e6 and a maximum fill time of 50 ms over a scan range of 300-1500 m/z. For MS2 resolution was set to 7500, isolation window was set to 1 m/z, normalized collision energy was set to 30% over a scan range of 150-1700 m/z and the AGC target was 1000% or 1×10e6 and a maximum fill time of 10 ms. Following the DDA scan the peptides were acquired using a unscheduled parallel reaction monitoring (PRM) strategy within the biological matrix to obtain retention times, using the following mass list for endogenous and reference heavy peptides: DINNIDY[+80]Y[+80]K (659.243634 m/z), DINNIDY[+80]Y[+80]K[+8] (663.250733 m/z), DINNIDYYK (579.277303 m/z), DINNIDYYK[+8] (583.284402 m/z), EIQNGNLHES[+80]DSESVPR (995.92874 m/z), EIQNGNL-HES[+80]DSESVPR[+10] (1000.932875 m/z), EIQNGNLHESDSESVPR (955.945575 m/z), EIQNGNLHESDSESVPR[+10] (960.949709 m/z), GISSLPR (365.21632 m/z), GISSLPR[+10] (370.220455 m/z), GISS[+80]LPR (405.199486 m/z), and GISS[+80]LPR[+10] (410.20362 m/z). For MS1, resolution was set to 120000 with an AGC target of 300% or 3×10e6 and a maximum fill time of 246 ms over a scan range of 300-1500 m/z. For MS2 resolution was set to 7500, isolation window was set to 1 m/z, normalized collision energy was set to 27% over a scan range of 150-1700 m/z and the AGC target was 1000% or 1×10e6 and a maximum fill time was set to auto. Raw data were analyzed via Skyline 49 (Version 24.1.0.414, Seattle, WA). The DDA survey run was imported and converted into a spectral library. The spectral library was built using the directed DDA runs, including following modifications, Carbamidomethyl (C), phosphorylation (STY) and isotope modifcations at the C-terminus (13C(6)15N(4) (C-term R) and 13C(6)15N(2) (C-term K)). Light and heavy peptides were included and unscheduled runs were used to detect retention times that were exported and used for scheduled PRM runs. MS1 filtering was performed by including a minimum of 20% of the base peak at 10 ppm accuracy. MS2 filtering was set to PRM as acquisition method. Data Availability The InstaNovo base model has been trained on the ProteomeTools datasets, part I-III and can be found in the ProteomeXchange Consortium via the PRIDE partner repository 29 with identifiers PXD004732 (Part I), PXD010595 (Part II) and PXD021013 (Part III). The 21-PTM dataset (ProteomeTools part V) can also be found in PRIDE under the identifier PXD009449. The FGFR2 dataset has been deposited in PRIDE under dataset identifier PXD062859. Reviewers can access the FGFR2 dataset with username reviewer_pxd062859{at}ebi.ac.uk and password eZorng4dZASn. The targeted proteomics experiment files have been deposited to PanoramaWeb and can be accessed with the link https://panoramaweb.org/4W3hr8.url with reviewer email panorama+reviewer341{at}proteinms.net and password 4^+NDURy%8H!Al. The InstaNovo-P training dataset can be found on HuggingFace at https://huggingface.co/datasets/InstaDeepAI/InstaNovo-P . Supplementary data supporting the data preprocessing and analysis performed on our training and validation datasets can be found on figshare at 10.6084/m9.figshare.29049533. Code Availability Inference code and model checkpoints for the base InstaNovo model are available at the InstaNovo GitHub repository https://github.com/instadeepai/instanovo . The checkpoint for InstaNovo-P v1.0.0 is available as an asset to InstaNovo release v1.1.2 as well as a Jupyter notebook describing the usage of InstaNovo-P . Code used for fine-tuning will be shared upon publication. Custom scripts used for data analysis and visualization are available upon request and will also be uploaded to a public repository upon publication. Author contributions statement K.K. and T.P.J. conceived the project. P.R., T.C, S.v.P, L.M. and C.F. provided datasets for training and validation. J.L. preprocessed the data, trained the model, and performed benchmarking with feedback from R.C., A.M., K.E., J.V.G, E.M.S., T.P.J and K.K. P.F. and J.F. performed cell culture, FGFR2 depletion, sample preparation and mass spectrometry, and C.F. searched the proteomics raw data. J.L. used the trained model to perform de novo sequencing in the validation datasets. A.K.M. and I.S.G. performed quantification of the FGFR2 dataset de novo predictions. V.C., C.F. and K.K. analysed the FGFR2 data. K.K. and J.L. performed proteome mapping and downstream analysis. N.L.C., J.V.G., T.P.J. and K.K. supervised the project. J.L., P.R., R.C., T.P.J. and K.K. drafted the original manuscript. All authors reviewed the manuscript. Additional information Competing interests R.C, A.M., K.E, N.L.C., and J.V.G. are employees of InstaDeep, 5 Merchant Square, London, UK. The other authors declare no competing interests. Supplementary Tables View this table: View inline View popup Download powerpoint Table 1. PRIDE Accession codes used for training, validation and test sets View this table: View inline View popup Download powerpoint Table 2. InstaNovo-P zero-shot evaluation on 3 datasets View this table: View inline View popup Download powerpoint Table 3. Peptide recall- and precision for peptides containing phosphorylation at different amino acids View this table: View inline View popup Download powerpoint Table 4. Hyperparameters Supplementary Figures Download figure Open in new tab Figure S1. Supplementary figure 1 . A) Learning rate during base model finetuning and InstaNovo-P training. B) Proportion of phosphorylation events per residue type in the FGFR2 and C) the 21PTM dataset. Download figure Open in new tab Figure S2. Supplementary figure 2 . A) Proportion of PSMs with 0 (blue), 1 (orange), and 2 phosphorylated sites (green) in the 3 test datasets used. B) Computed confidence thresholds for different FDR in our 3 datasets. Download figure Open in new tab Figure S3. Supplementary figure 3 . A) Confidence distribution for peptide predictions of InstaNovo-P in our test dataset, B) the 21PTM dataset, and C) the FGFR2 dataset. Download figure Open in new tab Figure S4. Supplementary figure 4 . Mapping of phosphorylated peptides identified with database search (bottom), and novel phosphorylated peptides identified with InstaNovo-P in LMNA (Uniprot accession P02545 ) at 5% FDR. Download figure Open in new tab Figure S5. Supplementary figure 5 . Quantification of semi-tryptic phosphorylated and non-phosphorylated peptides detected with InstaNovo-P in protein NPM1 (Uniprot accession P06748 ) at 5% FDR. Acknowledgements K.K. is supported by a Novo Nordisk Foundation Young Investigator Award (grant no. NNF16OC0020670) and a postdoctoral fellowship grant from the Independent Research Fund Denmark (grant no. 4257-00010B). K.K. and E.M.S. also acknowledge support from PRO-MS: Danish National Mass Spectrometry Platform for Functional Proteomics (grant no. 5072-00007B). C.F., J.F., and P.F. are supported by Wellcome Trust (107636/ Z/15/Z and 107636/Z/15/A). C.F. is also supported by a Novo Nordisk Foundation Young Investigator Award (grant no. NNF22OC0070845). V.C. is supported by the LEO Foundation (LF-OC-23-001220 to C.F.). We are grateful to the DTU Proteomics Core Facility for maintenance and running of instruments. We especially thank M. Wennekers Nielsen and M. Vestegaard Lukassen for their help and advice during troubleshooting, sample preparation and data acquisition. Footnotes https://figshare.com/articles/dataset/InstaNovo-P_supplementary_data/29049533 https://huggingface.co/datasets/InstaDeepAI/InstaNovo-P References 1. ↵ Pawson , T. & Scott , J. D. Protein phosphorylation in signaling – 50 years and counting . Trends Biochem. Sci . 30 , 286 – 290 , DOI: 10.1016/j.tibs.2005.04.013 ( 2005 ). Publisher: Elsevier . OpenUrl CrossRef PubMed Web of Science 2. Ubersax , J. A. & Ferrell Jr , J. E. Mechanisms of specificity in protein phosphorylation . Nat. Rev. Mol. Cell Biol . 8 , 530 – 541 , DOI: 10.1038/nrm2203 ( 2007 ). OpenUrl CrossRef PubMed Web of Science 3. ↵ Macek , B. , Mann , M. & Olsen , J. V. Global and site-specific quantitative phosphoproteomics: principles and applications . Annu. review pharmacology toxicology 49 , 199 – 221 , DOI: 10.1146/annurev.pharmtox.011008.145606 ( 2009 ). Place: United States . OpenUrl CrossRef PubMed Web of Science 4. ↵ Ardito , F. , Giuliani , M. , Perrone , D. , Troiano , G. & Lo Muzio , L. The crucial role of protein phosphorylation in cell signaling and its use as targeted therapy (Review ). Int. journal molecular medicine 40 , 271 – 280 , DOI: 10.3892/ijmm.2017.3036 ( 2017 ). Place: Greece . OpenUrl CrossRef PubMed 5. ↵ Castelo-Soccio , L. et al. Protein kinases: drug targets for immunological disorders . Nat. Rev. Immunol . 23 , 787 – 806 , DOI: 10.1038/s41577-023-00877-7 ( 2023 ). OpenUrl CrossRef 6. ↵ Sharma , K. et al. Ultradeep human phosphoproteome reveals a distinct regulatory nature of Tyr and Ser/Thr-based signaling . Cell reports 8 , 1583 – 1594 , DOI: 10.1016/j.celrep.2014.07.036 ( 2014 ). Place: United States . OpenUrl CrossRef PubMed Web of Science 7. ↵ Attwood , M. M. , Fabbro , D. , Sokolov , A. V. , Knapp , S. & Schiöth , H. B. Trends in kinase drug discovery: targets, indications and inhibitor design . Nat. Rev. Drug Discov . 20 , 839 – 861 , DOI: 10.1038/s41573-021-00252-y ( 2021 ). OpenUrl CrossRef PubMed 8. ↵ Bhullar , K. S. et al. Kinase-targeted cancer therapies: progress, challenges and future directions . Mol. Cancer 17 , 48 , DOI: 10.1186/s12943-018-0804-2 ( 2018 ). OpenUrl CrossRef PubMed 9. ↵ Olsen , J. V. & Mann , M. Status of large-scale analysis of post-translational modifications by mass spectrometry . Mol. & cellular proteomics : MCP 12 , 3444 – 3452 , DOI: 10.1074/mcp.O113.034181 ( 2013 ). Place: United States . OpenUrl Abstract / FREE Full Text 10. ↵ Hornbeck , P. V. et al. PhosphoSitePlus, 2014: mutations, PTMs and recalibrations . Nucleic Acids Res . 43 , D512 – D520 , DOI: 10.1093/nar/gku1267 ( 2014 ). _eprint: https://academic.oup.com/nar/article-pdf/43/D1/D512/17437800/gku1267.pdf . OpenUrl CrossRef PubMed 11. Ficarro , S. B. et al. Phosphoproteome analysis by mass spectrometry and its application to Saccharomyces cerevisiae . Nat. biotechnology 20 , 301 – 305 , DOI: 10.1038/nbt0302-301 ( 2002 ). Place: United States . OpenUrl CrossRef PubMed Web of Science 12. Olsen , J. V. et al. Quantitative phosphoproteomics reveals widespread full phosphorylation site occupancy during mitosis . Sci. signaling 3 , ra3 , DOI: 10.1126/scisignal.2000475 ( 2010 ). Place: United States . OpenUrl Abstract / FREE Full Text 13. ↵ Francavilla , C. et al. Multilayered proteomics reveals molecular switches dictating ligand-dependent EGFR trafficking . Nat. Struct. & Mol. Biol . 23 , 608 – 618 , DOI: 10.1038/nsmb.3218 ( 2016 ). OpenUrl CrossRef PubMed 14. ↵ Jedrychowski , M. P. et al. Evaluation of HCD- and CID-type fragmentation within their respective detection platforms for murine phosphoproteomics . Mol. & cellular proteomics : MCP 10 , M111.009910 , DOI: 10.1074/mcp.M111.009910 ( 2011 ). Place: United States . OpenUrl Abstract / FREE Full Text 15. Larsen , M. R. , Graham , M. E. , Robinson , P. J. & Roepstorff , P. Improved detection of hydrophilic phosphopeptides using graphite powder microcolumns and mass spectrometry: evidence for in vivo doubly phosphorylated dynamin I and dynamin III . Mol. & cellular proteomics : MCP 3 , 456 – 465 , DOI: 10.1074/mcp.M300105-MCP200 ( 2004 ). Place: United States . OpenUrl Abstract / FREE Full Text 16. Rush , J. et al. Immunoaffinity profiling of tyrosine phosphorylation in cancer cells . Nat. biotechnology 23 , 94 – 101 , DOI: 10.1038/nbt1046 ( 2005 ). Place: United States . OpenUrl CrossRef PubMed Web of Science 17. Blagoev , B. et al. A proteomics strategy to elucidate functional protein-protein interactions applied to EGF signaling . Nat. biotechnology 21 , 315 – 318 , DOI: 10.1038/nbt790 ( 2003 ). Place: United States . OpenUrl CrossRef PubMed Web of Science 18. Larsen , M. R. , Thingholm , T. E. , Jensen , O. N. , Roepstorff , P. & Jørgensen , T. J. D. Highly selective enrichment of phosphorylated peptides from peptide mixtures using titanium dioxide microcolumns . Mol. & cellular proteomics : MCP 4 , 873 – 886 , DOI: 10.1074/mcp.T500007-MCP200 ( 2005 ). Place: United States . OpenUrl Abstract / FREE Full Text 19. ↵ Watson , J. et al. Spatially resolved phosphoproteomics reveals fibroblast growth factor receptor recycling-driven regulation of autophagy and survival . Nat. Commun . 13 , 6589 , DOI: 10.1038/s41467-022-34298-2 ( 2022 ). OpenUrl CrossRef PubMed 20. ↵ Olsen , J. V. et al. Higher-energy C-trap dissociation for peptide modification analysis . Nat. methods 4 , 709 – 712 , DOI: 10.1038/nmeth1060 ( 2007 ). Place: United States . OpenUrl CrossRef PubMed Web of Science 21. ↵ Beausoleil , S. A. , Villén , J. , Gerber , S. A. , Rush , J. & Gygi , S. P. A probability-based approach for high-throughput protein phosphorylation analysis and site localization . Nat. biotechnology 24 , 1285 – 1292 , DOI: 10.1038/nbt1240 ( 2006 ). Place: United States . OpenUrl CrossRef PubMed Web of Science 22. Hjerrild , M. et al. Identification of phosphorylation sites in protein kinase A substrates using artificial neural networks and mass spectrometry . J. proteome research 3 , 426 – 433 , DOI: 10.1021/pr0341033 ( 2004 ). Place: United States . OpenUrl CrossRef PubMed Web of Science 23. ↵ Yu , F. et al. Identification of modified peptides using localization-aware open search . Nat. Commun . 11 , 4065 , DOI: 10.1038/s41467-020-17921-y ( 2020 ). OpenUrl CrossRef PubMed 24. ↵ Eloff , K. et al. InstaNovo enables diffusion-powered de novo peptide sequencing in large-scale proteomics experiments . Nat. Mach. Intell . DOI: 10.1038/s42256-025-01019-5 ( 2025 ). OpenUrl CrossRef 25. Yilmaz , M. et al. Sequence-to-sequence translation from mass spectra to peptides with a transformer model . Nat. Commun . 15 , 6427 , DOI: 10.1038/s41467-024-49731-x ( 2024 ). OpenUrl CrossRef PubMed 26. ↵ Tran , N. H. , Zhang , X. , Xin , L. , Shan , B. & Li , M. De novo peptide sequencing by deep learning . Proc. Natl. Acad. Sci. United States Am . 114 , 8247 – 8252 , DOI: 10.1073/pnas.1705691114 ( 2017 ). Place: United States . OpenUrl Abstract / FREE Full Text 27. ↵ Xia , J. et al. Adanovo: Adaptive {De Novo} peptide sequencing with conditional mutual information . ArXiv abs/2403.07013 ( 2024 ). 28. ↵ Zhang , X. et al. -primenovo: an accurate and efficient non-autoregressive deep learning model for de novo peptide sequencing . Nat. Commun . 16 , DOI: 10.1038/s41467-024-55021-3 ( 2025 ). OpenUrl CrossRef PubMed 29. ↵ Perez-Riverol , Y. et al. The PRIDE database at 20 years: 2025 update . Nucleic acids research 53 , D543 – D553 , DOI: 10.1093/nar/gkae1011 ( 2025 ). Place: England . OpenUrl CrossRef PubMed 30. ↵ Zolg , D. P. et al. Building ProteomeTools based on a complete synthetic human proteome . Nat. Methods 14 , 259 – 262 , DOI: 10.1038/nmeth.4153 ( 2017 ). OpenUrl CrossRef 31. ↵ Ramasamy , P. et al. Scop3p: A comprehensive resource of human phosphosites within their full context . J. Proteome Res . 19 , 3478 – 3486 , DOI: 10.1021/acs.jproteome.0c00306 ( 2020 ). PMID: 32508104 , 10.1021/acs.jproteome.0c00306. OpenUrl CrossRef PubMed 32. ↵ Teufel , F. et al. GraphPart: homology partitioning for biological sequence analysis . NAR Genomics Bioinforma . 5 , qad088 , DOI: 10.1093/nargab/lqad088 ( 2023 ). https://academic.oup.com/nargab/article-pdf/5/4/lqad088/52174155/lqad088.pdf . OpenUrl CrossRef 33. ↵ Howard , J. & Ruder , S. Universal language model fine-tuning for text classification . arXiv preprint arxiv: 1801.06146 ( 2018 ). 34. ↵ Zolg , D. P. et al. ProteomeTools: Systematic characterization of 21 post-translational protein modifications by liquid chromatography tandem mass spectrometry (LC-MS/MS) using synthetic peptides . Mol. Cell. Proteomics 17 , 1850 – 1863 ( 2018 ). OpenUrl Abstract / FREE Full Text 35. ↵ Wang , H. , Guo , M. , Wei , H. & Chen , Y. Targeting p53 pathways: mechanisms, structures and advances in therapy . Signal Transduct. Target. Ther . 8 , 92 , DOI: 10.1038/s41392-023-01347-1 ( 2023 ). OpenUrl CrossRef PubMed 36. ↵ Wang , Y. et al. Lamin A/C-dependent chromatin architecture safeguards naïve pluripotency to prevent aberrant cardiovascular cell fate and function . Nat. Commun . 13 , 6663 , DOI: 10.1038/s41467-022-34366-7 ( 2022 ). OpenUrl CrossRef 37. ↵ Ornitz , D. M. & Itoh , N. New developments in the biology of fibroblast growth factors . WIREs mechanisms disease 14 , e1549 , DOI: 10.1002/wsbm.1549 ( 2022 ). Place: United States. OpenUrl CrossRef 38. ↵ Shandilya , J. et al. Phosphorylation of multifunctional nucleolar protein nucleophosmin (NPM1) by aurora kinase B is critical for mitotic progression . FEBS Lett . 588 , 2198 – 2205 , DOI: 10.1016/j.febslet.2014.05.014 ( 2014 ). Publisher: John Wiley & Sons, Ltd . OpenUrl CrossRef PubMed 39. ↵ Mitrea , D. M. et al. Structural polymorphism in the N-terminal oligomerization domain of NPM1 . Proc. Natl. Acad. Sci . 111 , 4466 – 4471 , DOI: 10.1073/pnas.1321007111 ( 2014 ). Publisher: Proceedings of the National Academy of Sciences . OpenUrl Abstract / FREE Full Text 40. ↵ Guery , L. et al. Fine-tuning nucleophosmin in macrophage differentiation and activation . Blood 118 , 4694 – 4704 , DOI: 10.1182/blood-2011-03-341255 ( 2011 ). OpenUrl Abstract / FREE Full Text 41. ↵ Zhou , Y. et al. Metascape provides a biologist-oriented resource for the analysis of systems-level datasets . Nat. communications 10 , 1523 , DOI: 10.1038/s41467-019-09234-6 ( 2019 ). Place: England . OpenUrl CrossRef PubMed 42. ↵ Smith , M. P. et al. Reciprocal priming between receptor tyrosine kinases at recycling endosomes orchestrates cellular signalling outputs . The EMBO journal 40 , e107182 , DOI: 10.15252/embj.2020107182 ( 2021 ). Place: England . OpenUrl CrossRef 43. Francavilla , C. & O’Brien , C. S. Fibroblast growth factor receptor signalling dysregulation and targeting in breast cancer . Open biology 12 , 210373 , DOI: 10.1098/rsob.210373 ( 2022 ). Place: England . OpenUrl CrossRef PubMed 44. ↵ Francavilla , C. et al. Functional proteomics defines the molecular switch underlying FGF receptor trafficking and cellular outputs . Mol. cell 51 , 707 – 722 , DOI: 10.1016/j.molcel.2013.08.002 ( 2013 ). Place: United States . OpenUrl CrossRef PubMed Web of Science 45. ↵ Steinegger , M. & Söding , J. Mmseqs2 enables sensitive protein sequence searching for the analysis of massive data sets . Nat. Biotechnol . 35 , 1026 – 1028 , DOI: 10.1038/nbt.3988 ( 2017 ). OpenUrl CrossRef PubMed 46. ↵ Ansel , J. et al. PyTorch 2: Faster Machine Learning Through Dynamic Python Bytecode Transformation and Graph Compilation . In 29th ACM International Conference on Architectural Support for Programming Languages and Operating Systems, Volume 2 (ASPLOS ‘24) , DOI: 10.1145/3620665.3640366 ( ACM , 2024 ). OpenUrl CrossRef 47. ↵ Falcon , W. & The PyTorch Lightning team . PyTorch Lightning , DOI: 10.5281/zenodo.3828935 ( 2019 ). OpenUrl CrossRef 48. ↵ Dale , D. Fine-Tuning Scheduler , DOI: 10.5281/zenodo.6463952 ( 2022 ). OpenUrl CrossRef 49. ↵ MacLean , B. et al. Skyline: an open source document editor for creating and analyzing targeted proteomics experiments . Bioinforma. (Oxford, England) 26 , 966 – 968 , DOI: 10.1093/bioinformatics/btq054 ( 2010 ). Place: England . OpenUrl CrossRef PubMed Web of Science 50. Peng , X. et al. Identification of missing proteins in the phosphoproteome of kidney cancer . J. Proteome Res . 16 , 4364 – 4373 , DOI: 10.1021/acs.jproteome.7b00332 ( 2017 ). OpenUrl CrossRef PubMed 51. Post , H. et al. Robust, sensitive, and automated phosphopeptide enrichment optimized for low sample amounts applied to primary hippocampal neurons . J. Proteome Res . 16 , 728 – 737 , DOI: 10.1021/acs.jproteome.6b00753 ( 2016 ). OpenUrl CrossRef PubMed 52. Tsiatsiani , L. et al. Opposite electron-transfer dissociation and higher-energy collisional dissociation fragmentation characteristics of proteolytic k/r(x)n and (x)nk/r peptides provide benefits for peptide sequencing in proteomics and phosphoproteomics . J. Proteome Res . 16 , 852 – 861 , DOI: 10.1021/acs.jproteome.6b00825 ( 2016 ). OpenUrl CrossRef PubMed 53. Espadas , G. , Borràs , E. , Chiva , C. & Sabidó , E. Evaluation of different peptide fragmentation types and mass analyzers in data-dependent methods using an orbitrap fusion lumos tribrid mass spectrometer . PROTEOMICS 17 , DOI: 10.1002/pmic.201600416 ( 2017 ). OpenUrl CrossRef 54. Bekker-Jensen , D. B. et al. An optimized shotgun strategy for the rapid generation of comprehensive human proteomes . Cell Syst . 4 , 587 – 599 .e4 , DOI: 10.1016/j.cels.2017.05.009 ( 2017 ). OpenUrl CrossRef PubMed 55. Tran , T. T. , Strozynski , M. & Thiede , B. Quantitative phosphoproteome analysis of cisplatin-induced apoptosis in jurkat t cells . PROTEOMICS 17 , DOI: 10.1002/pmic.201600470 ( 2017 ). OpenUrl CrossRef 56. Liu , Z. , Wang , F. , Chen , J. , Zhou , Y. & Zou , H. Modulating the selectivity of affinity absorbents to multi-phosphopeptides by a competitive substitution strategy . J. Chromatogr. A 1461 , 35 – 41 , DOI: 10.1016/j.chroma.2016.07.042 ( 2016 ). OpenUrl CrossRef PubMed 57. Humphrey , E. S. et al. Resolution of novel pancreatic ductal adenocarcinoma subtypes by global phosphotyrosine profiling . Mol. Cell. Proteomics 15 , 2671 – 2685 ( 2016 ). OpenUrl Abstract / FREE Full Text 58. Picariello , G. et al. Antibody-independent identification of bovine milk-derived peptides in breast-milk . Food Funct . 7 , 3402 – 3409 ( 2016 ). OpenUrl CrossRef PubMed 59. Francavilla , C. et al. Phosphoproteomics of primary cells reveals druggable kinase signatures in ovarian cancer . Cell Reports 18 , 3242 – 3256 , DOI: 10.1016/j.celrep.2017.03.015 ( 2017 ). OpenUrl CrossRef PubMed 60. Lyon , S. M. et al. A method for whole protein isolation from human cranial bone . Anal. Biochem . 515 , 33 – 39 , DOI: 10.1016/j.ab.2016.09.021 ( 2016 ). OpenUrl CrossRef PubMed 61. Nguyen , E. V. et al. Hyper-phosphorylation of sequestosome-1 distinguishes resistance to cisplatin in patient derived high grade serous ovarian cancer cells . Mol. amp; Cell. Proteomics 16 , 1377 – 1392 , DOI: 10.1074/mcp.m116.058321 ( 2017 ). OpenUrl CrossRef 62. Drake , J. M. et al. Phosphoproteome integration reveals patient-specific networks in prostate cancer . Cell 166 , 1041 – 1054 , DOI: 10.1016/j.cell.2016.07.007 ( 2016 ). OpenUrl CrossRef PubMed 63. Su , N. et al. Special enrichment strategies greatly increase the efficiency of missing proteins identification from regular proteome samples . J. Proteome Res . 14 , 3680 – 3692 ( 2015 ). OpenUrl CrossRef PubMed 64. Creedon , H. et al. Identification of novel pathways linking epithelial-to-mesenchymal transition with resistance to her2-targeted therapy . Oncotarget 7 , 11539 – 11552 , DOI: 10.18632/oncotarget.7317 ( 2016 ). OpenUrl CrossRef PubMed 65. van der Mijn , J. C. et al. Evaluation of different phospho-tyrosine antibodies for label-free phosphoproteomics . J. Proteomics 127 , 259 – 263 , DOI: 10.1016/j.jprot.2015.04.006 ( 2015 ). OpenUrl CrossRef PubMed 66. Piersma , S. R. et al. Feasibility of label-free phosphoproteomics and application to base-line signaling of colorectal cancer cell lines . J. Proteomics 127 , 247 – 258 , DOI: 10.1016/j.jprot.2015.03.019 ( 2015 ). OpenUrl CrossRef PubMed 67. Ruprecht , B. et al. Comprehensive and reproducible phosphopeptide enrichment using iron immobilized metal ion affinity chromatography (fe-imac) columns . Mol. amp; Cell. Proteomics 14 , 205 – 215 , DOI: 10.1074/mcp.m114.043109 ( 2015 ). OpenUrl CrossRef 68. Kauko , O. et al. Label-free quantitative phosphoproteomics with novel pairwise abundance normalization reveals synergistic ras and cip2a signaling . Sci. Reports 5 , DOI: 10.1038/srep13099 ( 2015 ). OpenUrl CrossRef PubMed 69. Alpert , A. J. , Hudecz , O. & Mechtler , K. Anion-exchange chromatography of phosphopeptides: weak anion exchange versus strong anion exchange and anion-exchange chromatography versus electrostatic repulsion-hydrophilic interaction chromatography . Anal. Chem . 87 , 4704 – 4711 ( 2015 ). OpenUrl CrossRef 70. Suni , V. , Imanishi , S. Y. , Maiolica , A. , Aebersold , R. & Corthals , G. L. Confident site localization using a simulated phosphopeptide spectral library . J. Proteome Res . 14 , 2348 – 2359 ( 2015 ). OpenUrl CrossRef PubMed 71. Sharma , K. et al. Ultradeep human phosphoproteome reveals a distinct regulatory nature of tyr and ser/thr-based signaling . Cell Reports 8 , 1583 – 1594 , DOI: 10.1016/j.celrep.2014.07.036 ( 2014 ). OpenUrl CrossRef PubMed Web of Science 72. Tong , J. et al. Integrated analysis of proteome, phosphotyrosine-proteome, tyrosine-kinome, and tyrosine-phosphatome in acute myeloid leukemia . PROTEOMICS 17 , DOI: 10.1002/pmic.201600361 ( 2017 ). OpenUrl CrossRef 73. Bauer , M. et al. Evaluation of data-dependent and -independent mass spectrometric workflows for sensitive quantification of proteins and phosphorylation sites . J. Proteome Res . 13 , 5973 – 5988 ( 2014 ). OpenUrl CrossRef PubMed 74. Shevchuk , O. et al. HOPE-fixation of lung tissue allows retrospective proteome and phosphoproteome studies . J. Proteome Res . 13 , 5230 – 5239 ( 2014 ). OpenUrl CrossRef PubMed 75. Molden , R. C. , Goya , J. , Khan , Z. & Garcia , B. A. Stable isotope labeling of phosphoproteins for large-scale phosphorylation rate determination . Mol. amp; Cell. Proteomics 13 , 1106 – 1118 , DOI: 10.1074/mcp.o113.036145 ( 2014 ). OpenUrl CrossRef 76. Rajeeve , V. , Vendrell , I. , Wilkes , E. , Torbett , N. & Cutillas , P. R. Cross-species proteomics reveals specific modulation of signaling in cancer and stromal cells by phosphoinositide 3-kinase (PI3K) inhibitors . Mol. Cell. Proteomics 13 , 1457 – 1470 ( 2014 ). OpenUrl Abstract / FREE Full Text View the discussion thread. Back to top Previous Next Posted May 18, 2025. Download PDF Data/Code Email Thank you for your interest in spreading the word about bioRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following InstaNovo-P: A de novo peptide sequencing model for phosphoproteomics Message Subject (Your Name) has forwarded a page to you from bioRxiv Message Body (Your Name) thought you would like to see this page from the bioRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share InstaNovo-P: A de novo peptide sequencing model for phosphoproteomics Jesper Lauridsen , Pathmanaban Ramasamy , Rachel Catzel , Vahap Canbay , Amandla Mabona , Kevin Eloff , Paul Fullwood , Jennifer Ferguson , Annekatrine Kirketerp-Møller , Ida Sofie Goldschmidt , Tine Claeys , Sam van Puyenbroeck , Nicolas Lopez Carranza , Erwin M. Schoof , Lennart Martens , Jeroen Van Goey , Chiara Francavilla , Timothy Patrick Jenkins , Konstantinos Kalogeropoulos bioRxiv 2025.05.14.654049; doi: https://doi.org/10.1101/2025.05.14.654049 Share This Article: Copy Citation Tools InstaNovo-P: A de novo peptide sequencing model for phosphoproteomics Jesper Lauridsen , Pathmanaban Ramasamy , Rachel Catzel , Vahap Canbay , Amandla Mabona , Kevin Eloff , Paul Fullwood , Jennifer Ferguson , Annekatrine Kirketerp-Møller , Ida Sofie Goldschmidt , Tine Claeys , Sam van Puyenbroeck , Nicolas Lopez Carranza , Erwin M. Schoof , Lennart Martens , Jeroen Van Goey , Chiara Francavilla , Timothy Patrick Jenkins , Konstantinos Kalogeropoulos bioRxiv 2025.05.14.654049; doi: https://doi.org/10.1101/2025.05.14.654049 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Bioinformatics Subject Areas All Articles Animal Behavior and Cognition (7637) Biochemistry (17705) Bioengineering (13899) Bioinformatics (41968) Biophysics (21460) Cancer Biology (18603) Cell Biology (25526) Clinical Trials (138) Developmental Biology (13385) Ecology (19910) Epidemiology (2067) Evolutionary Biology (24328) Genetics (15614) Genomics (22513) Immunology (17741) Microbiology (40423) Molecular Biology (17193) Neuroscience (88646) Paleontology (667) Pathology (2835) Pharmacology and Toxicology (4827) Physiology (7647) Plant Biology (15160) Scientific Communication and Education (2046) Synthetic Biology (4302) Systems Biology (9825) Zoology (2271)
Text is read by the "Ask this paper" AI Q&A widget below.
Extraction quality varies by source — PMC NXML preserves structure
cleanly, OA-HTML may include some navigation residue, and OA-PDF can
have broken hyphenation. The publisher copy
(via DOI)
is the canonical version.