Context-based protein function prediction in bacterial genomes

preprint OA: closed
📄 Open PDF Full text JSON View at publisher

Abstract

Motivation The rapid growth of sequencing data from high-throughput technologies has emphasized the need to uncover the functions of unannotated genes. Recent advancements in deep learning algorithms have enabled researchers to utilize various features to predict protein functions. Traditionally, these algorithms treat proteins as independent functional units or consider interactions only at the protein level. However, prokaryotes often preserve specific genomic neighborhoods over evolutionary time, providing valuable context for predicting protein functions. This context can arise from genes near the gene of interest or synteny regions, where the conserved order of genes on chromosomes results from common ancestry. Results We developed a transformer-based model to pre-train representations of proteins based on their genomic context, and use this model for predicting protein functions. Our results show that context-based protein representations capture context-specific functional semantics and can effectively predict protein functions. We use our model to investigate the influence of phylogenetic distance and homology on the performance of context-dependent function prediction, and find that synteny affects the prediction performance substantially, except for some functions where the function is determined by the genomic context. Our experiments allow us to gain insights into the factors affecting the performance and applicability of context-based function prediction methods across diverse prokaryotic genomes and meta-genomes. Availability and implementation The generated model, including all training code and generated data, is freely available at https://github.com/bio-ontology-research-group/Genomic_context . Contact [email protected]
Full text 50,827 characters · extracted from preprint-html · click to expand
Context-based protein function prediction in bacterial genomes | bioRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-M677548'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search New Results Context-based protein function prediction in bacterial genomes View ORCID Profile Daulet Toibazar , View ORCID Profile Maxat Kulmanov , View ORCID Profile Robert Hoehndorf doi: https://doi.org/10.1101/2024.10.14.618363 Daulet Toibazar 1 Bioengineering Program, Biological and Environmental Science and Engineering Division, King Abdullah University of Science and Technology (KAUST) , Thuwal, 23955, Makkah, Saudi Arabia Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Daulet Toibazar Maxat Kulmanov 2 Computer Science Program, Computer, Electrical and Mathematical Sciences & Engineering Division, King Abdullah University of Science and Technology (KAUST) , Thuwal, 23955, Makkah, Saudi Arabia Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Maxat Kulmanov Robert Hoehndorf 1 Bioengineering Program, Biological and Environmental Science and Engineering Division, King Abdullah University of Science and Technology (KAUST) , Thuwal, 23955, Makkah, Saudi Arabia 2 Computer Science Program, Computer, Electrical and Mathematical Sciences & Engineering Division, King Abdullah University of Science and Technology (KAUST) , Thuwal, 23955, Makkah, Saudi Arabia Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Robert Hoehndorf For correspondence: robert.hoehndorf{at}kaust.edu.sa Abstract Full Text Info/History Metrics Supplementary material Data/Code Preview PDF Abstract Motivation The rapid growth of sequencing data from high-throughput technologies has emphasized the need to uncover the functions of unannotated genes. Recent advancements in deep learning algorithms have enabled researchers to utilize various features to predict protein functions. Traditionally, these algorithms treat proteins as independent functional units or consider interactions only at the protein level. However, prokaryotes often preserve specific genomic neighborhoods over evolutionary time, providing valuable context for predicting protein functions. This context can arise from genes near the gene of interest or synteny regions, where the conserved order of genes on chromosomes results from common ancestry. Results We developed a transformer-based model to pre-train representations of proteins based on their genomic context, and use this model for predicting protein functions. Our results show that context-based protein representations capture context-specific functional semantics and can effectively predict protein functions. We use our model to investigate the influence of phylogenetic distance and homology on the performance of context-dependent function prediction, and find that synteny affects the prediction performance substantially, except for some functions where the function is determined by the genomic context. Our experiments allow us to gain insights into the factors affecting the performance and applicability of context-based function prediction methods across diverse prokaryotic genomes and meta-genomes. Availability and implementation The generated model, including all training code and generated data, is freely available at https://github.com/bio-ontology-research-group/Genomic_context . Contact robert.hoehndorf{at}kaust.edu.sa Introduction The rapid increase in genome and metagenome assemblies from diverse sources, driven by advancements in sequencing technologies and assembly methods have generated large amounts of genome and protein sequence data ( Bateman et al ., 2022 ; Delmont et al ., 2011 ; Parks et al ., 2017 ). Classical experimental methods to determine the function of proteins are highly informative ( Markwick et al ., 2008 ; Kemp and Alcock, 2017 ) but require substantial investment in terms of both time and materials, making them less accessible for large-scale or high-throughput studies. As a result, experimental function discovery has not kept pace with the rapid generation of new protein sequences, leading to a pronounced disparity between experimentally annotated and unannotated proteins. The sequence–structure–function paradigm states that the function of a protein is largely dependent on its structure, which in turn is dependent on the protein sequence. While state-of-the-art models like AlphaFold3 ( Abramson et al ., 2024 ) and ESMFold ( Lin et al ., 2023 ) have largely solved the challenge of predicting protein structure from sequence, predicting function from sequence or structure remains an open research question. Computational approaches leveraging deep learning algorithms have been employed to tackle this issue, using diverse input features such as protein sequence ( Kulmanov and Hoehndorf, 2019 ; Elhaj-Abdou et al ., 2021 ; Kulmanov et al ., 2021 ), protein–protein interaction (PPI) networks ( Nguyen et al ., 2011 ; Hou, 2017 ; Wan et al ., 2019 ), protein structure ( Konc et al ., 2013 ; Lai and Xu, 2021 ), or multiple modalities ( Cai et al ., 2020 ; You et al ., 2021 ). These methods show promising results in function prediction. However, they often treat proteins as independent functional units or only consider protein-level interactions. The arrangement of genes, in particular in prokaryotic genomes, is not random; co-regulated genes are assembled in operons for more effective gene regulation, and conserved genome regions (syntenic blocks) are preserved under selective pressure ( Rogozin, 2002 ; Guerrero et al ., 2005 ; Junier and Rivoire, 2013 ; Osbourn and Field, 2009 ). The information present in a gene’s genomic context has been used for protein function prediction in some of the first computational function prediction methods ( Huynen et al ., 2000 ; Wolf et al ., 2001 ). Novel machine learning algorithms are able to leverage genomic context information for discovering novel protein functions and novel protein interactions ( Miller et al ., 2022 ; Hwang et al ., 2024 ). The word2vec model ( Mikolov et al ., 2013 ), particularly, has been used for context-dependent function prediction ( Miller et al ., 2022 ) with a comparable performance to sequence-based baseline models for selected functions. However, the word2vec model aggregates different genomic context variations, resulting in a static representation of genes and overlooking alternative functions or participation of genes in distinct biological pathways, depending on their genomic context. A more recent study ( Hwang et al ., 2024 ) leverages the Bidirectional Encoder Representations from Transformers (BERT) ( Devlin et al ., 2018 ) to contextualize sequence-based Evolutionary Scale Modelling 2 (ESM2) ( Lin et al ., 2023 ) embeddings and find that contextualized protein embeddings encode biologically relevant information that relates to protein function and their interactions. However, machine learning models trained on genomic context will be affected by synteny between genomes and may memorize synteny blocks even in the absence of protein sequence similarity to effectively predict protein functions. Furthermore, previous models based on word2vec or transformers only evaluated a small set of manually selected functions based on prior hypotheses about which functions may be dependent on genomic context. To overcome these limitations, we developed a novel transformer-based model that exploits genomic context for protein function prediction. We test our model using InterPro IDs and a selected set of GO functions. We further evaluate the effects of phylogenetic relatedness and synteny on the performance of our context-based protein function prediction using our model. We demonstrate that our model can improve over prior models for context-based function prediction. Our model, including all training code and generated data, is freely available at https://github.com/bio-ontology-research-group/Genomic_context . Materials and methods Prokaryotic genome assemblies and gene clustering We downloaded a genome dataset from NCBI, available on January 30, 2023, ( https://www.ncbi.nlm.nih.gov/assembly ). The dataset comprised a total of 31,002 genomes containing 119,851,800 protein sequences. We clustered the 119,851,800 (putative) protein sequences using the MMseqs2 ( Steinegger and Söding, 2017 ) linclust algorithm, setting the min-seq-id parameter to 0.3. Clustering resulted in 8,988,233 clusters. We removed clusters with fewer than 20 sequences to further refine the dataset, eliminating approximately 18% of the original protein sequences. This filtration step reduced the number of clusters to 544,993 while retaining most initial sequences. We assigned a representative protein sequence to each cluster using mmseqs createsubdb DB_clu DB_DB_clu rep command. We used the InterProScan version 5.66-98.0 ( Jones et al ., 2014 ) to assign InterPro IDs to each cluster representative protein. These InterPro IDs serve as ground-truth labels for their corresponding protein sequences in further analysis. Reformulating genome corpus and creating MLM dataset Following Miller et al . (2022 ), we reformulated the genome corpus by substituting protein sequences with cluster-representative protein accession IDs. Replacing the protein sequences with protein IDs transforms the genomes into a “corpus” where tokens are proteins. We iterated through the genome corpus to generate the Masked Language Modeling (MLM) pre-training dataset, where all sentences consisted of 9 proteins. We divided the dataset into two parts: 90% for training and 10% for evaluation; we used the validation set to monitor the model’s training process and employed early stopping if the evaluation loss ceased to improve. BERT architecture and pre-training We developed a custom tokenizer using the HuggingFace ( Wolf et al ., 2019 ) tokenizer library with a vocabulary size of 544,998 (one for each cluster representative protein) to ensure each protein cluster is recognized as a separate token. We chose BERT as the base model for this project due to its bidirectional encoding strategy, which enables the comprehension of contextual information from input sentences. We used BERT with the following parameters: number hidden layers = 12, hidden size = 512, num attention heads = 8 . We pre-trained the BERT model de novo using the self-supervised MLM objective, where we randomly masked 20% of the input tokens, and the model predicts masked tokens based on the remaining tokens in the input sentence. The pre-training process used 8 V100 GPUs to accelerate the computations and enable efficient training on the large dataset. To streamline the training process, we used the HuggingFace (Wolf et al ., 2019) Trainer class. This class provides a high-level interface for training models, handling data loading, and managing the training loop. After pre-training, we evaluated the BERT with the “Fill-Mask” task, where we assessed the BERT’s ability to correctly predict the masked gene in a chunk of the genome using top-1 and top-5 accuracy. Top-1 accuracy takes into consideration the token with the highest predicted probability and compares it to ground truth, while top-5 accuracy takes into consideration the 5 highest prediction probability tokens . Extracting BERT embeddings and functional annotation We used our pre-trained BERT model to extract contextual protein representations by passing chunks of the genome: E = BERT( I ) (Supplementary Figure 1). Let I be the input to the model with n sentences and l representing the sentence length, I ∈ ℝ n ×l . From protein embeddings, denoted as E ∈ ℝ n×l×h , where h represents the embedding size, we extracted the embeddings for the proteins in the middle of each sentence. We annotated the middle protein to ensure balanced context-aware embeddings from both sides. Using proteins from different positions can yield varying embeddings due to slight shifts in the context window. Averaging embeddings in a sentence is redundant since each token embedding already contains contextual information about surrounding tokens. We obtained the corresponding labels for each x k ∈ ℝ h , k ∈ [1, 2, …, n ], using a multilabel binarized function from the sklearn library ( Pedregosa et al ., 2011 ). We presented the labels in a one-hot encoded form in a matrix Y ∈ ℝ n×c , where we define C as the set of all InterPro classes and | C | = c . In matrix Y, a value of 1 indicates that the given gene is annotated with a certain InterPro ID, and 0 is otherwise. For the supervised classification task, we trained a prediction MLP model with 3 hidden layers, configured with weights W i and biases b i , where i ∈ [0, 1, 2, 3]. The input layer accepts the embedding matrix X ∈ ℝ n×h , and the output layer produces Ŷ ∈ ℝ n×c . We used BCEWithLogitsLoss ( Paszke et al ., 2019 ) as the loss function to enable the multilabel classification objective. Additionally, we employed the Adam ( Kingma and Ba, 2014 ) optimizer with a learning rate of 0.001, which decayed by a gamma factor of 0.05 every 5 epochs. Incorporating phylogeny information into dataset splits To investigate the impact of evolutionary distance on contextual predictive performance, we divided the genome dataset into various train/test splits based on phylogeny ( de Vienne, 2016 ). Besides a random split, where all genomes were randomly split into train and test sets, we utilized splits at the family (e.g., Enterobacteriaceae ), order (e.g., Enterobacterales ), and class (e.g., Gammaproteobacteria ) levels. In each split, all genomes from a particular family were allocated to the test set, while the rest of the genomes outside of the specified phylogenetic circles were used to train the model (Supplementary Figure 2). We systematically widened the evolutionary gap between training and testing datasets to evaluate the model’s ability to capture and leverage genomic contextual cues beyond those confined to synteny regions or operons inherited from shared ancestors. We extracted BERT embeddings for all splits and trained an MLP classifier to annotate their corresponding InterPro IDs. We repeated this experiment using ESM2 embeddings derived from cluster-representative protein sequences to analyze the effect of evolutionary distance on sequence-based models and establish a baseline for our contextual approach. Specifically, we utilized the esm2 t30 150M UR50D model, featuring 30 layers and 640 embedding dimensions. Evaluation metrics We assess our experiments with two evaluation metrics: F max and AUC-ROC scores. The first metric is a protein-centric maximum value for the F score with threshold, t ∈ [0, 1]. We used the standard F max formulation used in the CAFA ( Radivojac et al ., 2013 ) challenge, where we define I t as a set of proteins annotated with at least one InterPro ( Eq. 1 ). Precision is averaged over proteins in I t ( Eq. 2 ), while recall is averaged over all proteins in the dataset ( Eq. 3 ). Then, F max is chosen from the maximum value of harmonic precision-recall mean over all possible values of t ( Eq. 4 ). The second evaluation metric is an InterPro term-centric AUC for the Receiver Operating Characteristic (ROC) curve. ROC curve shows the performance of our model at all classification thresholds. It plots the true-positive rate (TPR) against the false-positive rate (FPR). We calculate TPR and FPR for class j, j ∈ [1, 2, …, c ] with the following equations: Then, using TPR and FPR, we integrate over threshold values to find the AUC for the ROC curve. The final AUC value is calculated by taking micro and macro average of class-centric AUC values. Statistical analysis To assess the statistical significance of differences in F max values across various train/test splits, we conducted 4 trials for each experiment. Each trial used different random seeds for MLP model weight initialization and data shuffling, allowing us to add Gaussian noise to our experiments. We then used a one-way ANOVA ( Fisher, 1925 ) (Analysis of Variance) test to determine if there are statistically significant differences in the averaged F max between random, family, order, and class-level splits. If the ANOVA indicated significant differences between the groups, we conducted a post-hoc analysis using Tukey’s Honest Significant Difference (HSD) ( Tukey, 1949 ) test to identify which specific splits were statistically different. Extending function prediction to Gene Ontology We extended our protein function prediction task by incorporating experimental protein annotations from QuickGO ( https://www.ebi.ac.uk/QuickGO/annotations ) across three Gene Ontology (GO) branches: Molecular Function (MF), Biological Process (BP), and Cellular Component (CC). To annotate proteins in our corpus, we performed BLAST searches against these experimentally annotated proteins, using a sequence similarity threshold of > 90%. After completing the annotation process, we extracted BERT embeddings from proteins with experimental GO annotations. These embeddings served as input for training an MLP model, which learned to map them to corresponding GO functions. Term-centric AUC scores provided a metric for evaluating our prediction effectiveness. To focus our analysis, the results were filtered to highlight the top- n highest AUC scores along with their associated GO terms. Contextualizing these findings involved a performance comparison between our model and the state-of-the-art DeepGO-SE ( Kulmanov et al ., 2023 ) model. This comparative analysis aimed to explore the potential of genomic context information, as captured by BERT embeddings, in complementing the features utilized by the DeepGO-SE model. Investigating the generalization of BERT to diverse genomic architectures in prokaryotes Further, we compared our BERT model against the word2vec to investigate which model generalizes better to the diverse genomic contexts in bacterial genomes. For this comparison, we utilized the corpus proposed by Miller et al. ( Miller et al ., 2022 ), where they provide gene KEGG Orthology (KO) identifiers and categorize genes into functional classes. From all functional classes available in metadata, we selected the same 9 classes used in the original paper ( Miller et al ., 2022 ), excluding the ‘Others’ class. We extract all KO’s corresponding to 9 classes and obtain their context-dependent embeddings. For the BERT model, we pre-trained BERT using the MLM objective on the published genome corpus ( Miller et al ., 2022 ) and applied a similar strategy as shown in Supplementary Figure 1. To evaluate the word2vec model, we average the embeddings by considering 4 contextual proteins from both sides of the protein of interest, making the representation more context-specific (Supplementary Figure 3); otherwise, all embeddings from the same KO would be identical despite differences in the genomic context. Finally, we trained an MLP classifier that assigns a functional class to each embedding and evaluate the classification task using precision, recall, and F1 scores. We performed an evaluation with 3 repetitions, each time sampling 10,000 genomes from the corpus. Results Genomic language model: pre-training and evaluation We developed an encoder-based BERT model to assess if protein representation learning from genomic context suffices for protein function prediction in prokaryotes. Our method involved creating a consistent vocabulary across genomes by clustering proteins at 30% sequence identity and replacing proteins with their cluster’s accession identifier. We trained a BERT model using a masked language modeling objective, where 20% of the tokens are masked, and the model predicts them based on their genomic context. The objective of the training was to learn meaningful representations of proteins that capture their contextual relationships within the genomic regions where they appear. We conducted a “Fill-Mask” evaluation to assess the performance of the pre-trained BERT model. This method involves predicting a masked token within a sequence of tokens, allowing us to measure the model’s accuracy in identifying the correct token. The evaluation revealed a 94% top-1 accuracy and a 99% top-5 accuracy across 500 randomly selected genomes. Next, we predict protein functions using the proteins’ BERT embeddings and a prediction model. We divide our evaluation into three groups: general proteins (including proteins with enzymatic functions, such as synthases, dehydrogenases, ATPases, etc.), secretion proteins, and defense system proteins. We use the proteins’ InterPro identifiers as functional labels for each protein and design a prediction model to predict these functional labels from the BERT embeddings. The prediction model is represented by a multi-layer perception (MLP) model with 3 hidden layers and a ReLU activation function. Supplementary Figure 1 illustrates the workflow and how we perform gene function prediction. We benchmark context-dependent BERT embeddings against sequence-based ESM2 embeddings to assess the extent of genomic context information compared to protein sequence information for listed classes of bacterial proteins. Our results ( Table 1 ) show that defense system proteins achieve the highest F max value when predicted from BERT embeddings, followed by general and secretion proteins. On the other hand, predictions based on ESM2 embeddings achieve the highest F max for the general protein class, followed by secretion and defense proteins. Our findings show that, on average, the ESM-based model that uses protein sequences as input can better predict general protein function than it can predict functions related to the defense or secretion system, which are inherently context-dependent functions; the BERT-based model which exploits genomic context shows the opposite trend. Furthermore, we also find that BERT embeddings achieve higher F max and AUC values for all three types of proteins compared to ESM2 embeddings. View this table: View inline View popup Download powerpoint Table 1. Context-based function prediction performance in comparison to sequence-based ESM2 embeddings Evaluating the impact of phylogenetic distance on context-based function prediction performance We hypothesize that the increased performance observed for the context-based model comes from a combination of two phenomena: (1) functions that are inherently dependent on the genomic context (such as bacterial defense and secretion), and which the BERT model can exploit; and (2) synteny regions appearing in closely related organisms and which our MLP model memorizes when it sees the related genome during training. To test the extent to which defense and general protein classes’ genomic context information is influenced by synteny in closely related organisms, we designed an experiment to sequentially reduce the impact of synteny. In our experiment, we split the training and testing sets (consisting of entire genomes) according to their phylogenetic origin. Splitting organisms according to phylogenetic origin increases the evolutionary distance between the genomes in training and testing, and limits the ability of the MLP model to memorize genomic context. We systematically increase the phylogenetic distance from a random split to a split by family, order, and class levels. The bacterial families we analyzed include Enterobacteriaceae and Pseudomonadaceae , as these are the largest families in our genome dataset. At each phylogenetic level, we use genomes within the phylogenetic group as the test set; all genomes outside of the specified phylogenetic group are used as the training set (Supplementary Figure 2). The results of our experiments ( Table 2 ) demonstrate a substantial drop in performance as we increase the phylogenetic distance between training and testing sets. The drop in predictive performance from a random split to family, order, and class are all significant ( p ≈ 0.0); there is also a significant drop between order- and class-level splits ( p = 0.019). For general proteins, there is a significant difference in F max between random and any phylogeny-based split ( p ≈ 0.0) and between the family–class ( p = 0.002) and family–order ( p = 0.040) level splits. Supplementary Tables 1 and 2 contain detailed results of statistical analysis. View this table: View inline View popup Download powerpoint Table 2. Comparative analysis of F max scores for defense and general class proteins using BERT embeddings across phylogenetic distance Our results further demonstrate that defense proteins (which are known to be context-specific) achieve higher F max values in all phylogenetic splits when compared to general proteins (Supplementary Figure 4). Comparative study of phylogenetic distance on sequence-based models Models that use protein sequences should be less sensitive to increasing the evolutionary distance between training and testing, and should also be more accurate for general proteins. The hypothesis is based on the assumption that context-dependent systems, such as defense systems, often rely on genetic organization along with protein sequence for proper functioning ( Makarova et al ., 2011 ; Doron et al ., 2018 ). In contrast, general protein classes are expected to be less dependent (or independent) on contextual cues and more reliant on the intrinsic properties of the proteins themselves. Therefore, we repeat the experiment using the ESM2-based model ( Table 3 ). We observe that, as we increase the evolutionary distances from family over order to class for both Enterobacteriaceae and Pseudomonadaceae families, we observe a significant difference for defense proteins only when transitioning directly from a random split to a class-level split ( p = 0.036). A small decrease in mean F max values for defense proteins can likely be attributed to a higher protein sequence similarity between closely related organisms, and as we increase evolutionary distance, we decrease sequence similarity between the training and test sets for the MLP model. For general proteins, we observe no significant difference between any splits ( p ≥ 0.05). Furthermore, compared to defense proteins, general proteins achieve a higher F max across all splits. This observation aligns with our initial hypothesis that general proteins (e.g., enzymes) depend mainly on the protein’s intrinsic factors (i.e., sequence and structure) rather than genomic context. View this table: View inline View popup Download powerpoint Table 3. Comparative analysis of F max scores for defense and general class proteins using ESM2 embeddings across phylogenetic distance Context-only dependent Gene Ontology function prediction We extracted BERT embeddings from proteins with experimental GO annotations and trained an MLP model to predict GO functions from the embeddings. Our model’s performance was evaluated using term-centric AUC scores and compared against the state-of-the-art DeepGO-SE model. The terms that achieved the highest AUC scores from our BERT-based approach mostly showed AUC values close to 0.5 in DeepGO-SE predictions. Specifically: 17 out of 20 terms achieved higher AUC scores in BP, 15 out of 20 terms showed improved AUC values in MF, and 12 out of 15 terms exhibited better AUC scores in CC. The remaining terms performed similarly or slightly better with DeepGO-SE. Detailed results for each GO aspect are provided in Supplementary Tables 5, 6, and 7. One example of a function that our genomic context model can predict but DeepGO-SE cannot is “plasmid recombination” (GO:0042150). Plasmid recombination is influenced significantly by the genomic context, specifically the genes flanking the recombination sites on both the plasmid and the host genome. Homology between these flanking genes enhances the likelihood of homologous recombination, as sequence similarity provides the necessary substrate for the recombination machinery to align and exchange genetic material accurately (mol, 2022). Additionally, the presence of repetitive gene sequences, such as those generated by transposons, can drive non-homologous recombination, further emphasizing the role of surrounding genes in the process (mol, 2022). Given this dependency of this function on surrounding genes, it is plausible that the prediction of plasmid recombination as a protein function can be achieved more effectively from the genomic context than from sequence as done by DeepGO-SE. Our findings indicate that genomic context, as captured by BERT embeddings, provides valuable information for protein function prediction. This contextual information can be complementary to sequence-based information and therefore enhance the predictive power of existing models like DeepGO-SE. BERT generalizes better for different genomic contexts in bacterial genomes Our model is based on transformers; however, we could also have used (contextualized) word embeddings using a model such as word2vec instead. To determine whether the transformer model has advantages over word2vec, we pre-trained our BERT model on a corpus published by Miller et al. ( Miller et al ., 2022 ) and performed multi-class classification task using 9 classes presented in the original paper, except for the “Others” class. We also modified the word2vec classification pipeline to generate context-specific embeddings to ensure a fair comparison with BERT: we averaged word2vec embeddings across 9 genes, having 4 contextual genes from both sides. The results in Table 4 show that the transformer-based language model outperforms word2vec embeddings in F max score ( p = 4.652 · 10 − 9 ). The BERT model, and its attention heads, learn a more comprehensive and context-specific functional representations of genes and are better at leveraging contextual information within bacterial genomes than the contextualized word2vec model. The performance metrics per class for word2vec and BERT are given in Supplementary Tables 3 and 4, respectively. View this table: View inline View popup Download powerpoint Table 4. Functional classification results for BERT and contextualized word2vec embeddings. Discussion Effects of evolutionary distance and synteny Our study shows that protein functions can be predicted using only their genomic context as input, substantially improving predictive performance over models that use protein sequence (or their corresponding ESM2 embedding) as input at least for the classes of protein functions we evaluated (i.e., functions related to defense and secretion). Our findings also align with prior work that used word2vec ( Miller et al ., 2022 ) to predict these functions or that combined ESM2 embeddings with genomic context to form contextualized protein embeddings ( Hwang et al ., 2024 ). However, in contrast to prior work, we have developed a model that uses only the genomic context. Using our genomic context-based model, we were able to identify two distinct sources of information that are used to predict the protein functions: the genomic context, and evolutionary relatedness and synteny between genomes used in training and testing. The first aligns with our hypotheses and also the conclusions of prior work that finds genomic context to be important in predicting some protein functions; the second source of information, however, directly contradicts these conclusions. Exploiting similar or identical genomic arrangements between training and testing genomes (i.e., synteny) is usually undesirable as it severely limits the model’s ability to generalize, and results obtained by exploiting this information amount to a kind of “overfitting” of the model where it can perform well by only memorizing genomic arrangements seen during training. Therefore, we designed an experiment to evaluate the different protein categories under different phylogenetic distances – from random splits to family, order, and class-level splits; our experiment revealed that the genomic context models indeed exploit synteny. Synteny is due either to inheritance of the same regions from common ancestors or, in bacterial genomes, acquired through horizontal gene transfer. Our experiments provide an avenue to remove the effect of synteny from the prediction models. Such an approach is common in sequence-based function prediction where training proteins are usually expected to be sufficiently distinct from testing proteins (e.g., by employing a split based on sequence similarity). We also show that different functions are affected differently by synteny as well as by genomic context. Bacterial defense proteins can consistently be predicted better from genomic context than general proteins across all phylogenetic levels, indicating their functional predictability is less affected by phylogenetic distance. Furthermore, the prediction performance also declines significantly only when moving from random split to any phylogenetic split, whereas there is no significant difference when moving from family to order, or order to class. This finding supports our hypothesis that the predictive ability of contextual information is closely tied to conserved genomic regions and interactions with neighboring genes, i.e., that, when predicting defense mechanism functions, genomic context is more important than synteny, i.e., genomic context is what conveys the function to the proteins. We observe the opposite for general proteins with a pronounced and significant decay in predictive performance as contextual information increases. This decay suggests that the ability to predict protein function in general proteins is more sensitive to phylogenetic divergence, i.e., the performance of the BERT model primarily comes from synteny and exploiting similarity to genomic context seen for closely related genomes during training. Transformer generalize word2vec models for context-based function prediction Language models trained with self-supervised objectives have demonstrated state-of-the-art performance in many NLP tasks ( Raiaan et al ., 2024 ), proving their ability to understand language’s syntactic and semantic structure. In particular the attention mechanisms and the increased capacity of the model over word2vec representations can improve performance. In our study, the BERT model shows a superior performance compared to contextualized word2vec embeddings. However, improved predictive performance is not the only advantage of the BERT-based architecture. While the word2vec model can predict protein function based on context, it will output the same function for a protein independent of its context; however, proteins, even if they are sequence identical, can be involved in different functions in different genomes and, therefore, in different genomic contexts. Recent studies have shown that the CRISPR-Cas system, initially known for its role in bacterial immunity, is involved in functions beyond this primary purpose ( Devi et al ., 2022 ). We conducted a case study on the branch family of Cas2 protein (IPR010152) to visualize the contextual embeddings this protein generates in different bacterial genomes. Our findings revealed that there is one cluster located far from the other clusters, suggesting a function beyond adaptive immunity (Supplementary Figure 5). Conclusion We developed a model for protein function prediction based on genomic context. Our model is based on transformers and trained using a self-supervised masked language modeling objective. We demonstrate that our model improves over state-of-the-art models for context-based protein function predictions that are based on word2vec. Our model is also the first transformer model that is solely based on genomic context instead of combining genomic context with sequence-based representations. This separation allows us to demonstrate a kind of overfitting in context-based function prediction where the models memorize synteny between different genomes instead of actually learning the “semantics” of arrangements of genes in a genome and recognizing where this arrangement functions; it also allows us to remove or reduce this overfitting to synteny. Our model, together with all its training code, is freely available, allowing other researchers to build on and extend it. Limitations and future research In the future, we aim to leverage the Gene Ontology (GO) or KEGG functions more substantially because they offer more informative functional categories; for example, GO annotations encompass biological processes, molecular functions, and cellular components, offering a broader and more granular view of protein functions. To further advance this field of research, future studies focus on attention scores in different genome regions, which can open new insights into genetic interactions and operons. This approach would provide a more comprehensive understanding of the biological processes in which the gene is involved and the functional units of which they are a part. Another avenue for future research is the analysis of context-dependent function annotation and the effect of evolutionary relatedness in eukaryotic genomes, which have been reported to possess macro- and microsynteny structures ( Li et al ., 2022 ; Simakov et al ., 2022 ). Investigating the impact of genomic context on protein function prediction in eukaryotes could reveal novel insights into the evolutionary conservation and divergence of functional modules across different domains of life. Competing interests No competing interest is declared. Author contributions statement D.T. and R.H. conceived the experiments, D.T. conducted the experiments, D.T., M.K. and R.H. analyzed the results. D.T., R.H., and M.K. wrote and reviewed the manuscript. M.K. and R.H. obtained funding and supervised the research. All authors have read and approved the final version of the manuscript. Acknowledgments This work has been supported by funding from King Abdullah University of Science and Technology (KAUST) Office of Sponsored Research (OSR) under Award No. URF/1/4675-01-01, URF/1/4697-01-01, URF/1/5041-01-01, REI/1/5659-01-01, REI/1/5235-01-01, and FCC/1/1976-46-01. We acknowledge support from the KAUST Supercomputing Laboratory. Footnotes https://github.com/bio-ontology-research-group/Genomic_context References ( 2022 ). Molecular Biotechnology: Principles and Applications of Recombinant DNA . Wiley , Hoboken, New Jersey, USA, 6th edition . ↵ Abramson , J. et al. ( 2024 ). Accurate structure prediction of biomolecular interactions with alphafold 3 . Nature , 630 ( 8016 ), 493 – 500 . OpenUrl CrossRef PubMed ↵ Bateman , A. et al. ( 2022 ). Uniprot: the universal protein knowledgebase in 2023 . Nucleic Acids Research , 51 ( D1 ), D523 – D531 . OpenUrl ↵ Cai , Y. et al. ( 2020 ). Sdn2go: An integrated deep learning model for protein function prediction . Frontiers in Bioengineering and Biotechnology , 8 . ↵ de Vienne , D. M. ( 2016 ). Lifemap: Exploring the entire tree of life . PLOS Biology , 14 ( 12 ), e2001624 . OpenUrl CrossRef PubMed ↵ Delmont , T. O. et al. ( 2011 ). Metagenomic mining for microbiologists . The ISME Journal , 5 ( 12 ), 1837 – 1843 . OpenUrl CrossRef PubMed ↵ Devi , V. et al. ( 2022 ). Crispr-cas systems: role in cellular processes beyond adaptive immunity . Folia Microbiologica , 67 ( 6 ), 837 – 850 . OpenUrl CrossRef ↵ Devlin , J. et al. ( 2018 ). Bert: Pre-training of deep bidirectional transformers for language understanding . ↵ Doron , S. et al. ( 2018 ). Systematic discovery of antiphage defense systems in the microbial pangenome . Science , 359 ( 6379 ). ↵ Elhaj-Abdou , M. E. et al. ( 2021 ). Deep cnn lstm go: Protein function prediction from amino-acid sequences . Computational Biology and Chemistry , 95 , 107584 . OpenUrl CrossRef PubMed ↵ Fisher , R. A. ( 1925 ). Statistical Methods for Research Workers . Oliver and Boyd, Edinburgh . ↵ Guerrero , G. et al. ( 2005 ). Evolutionary, structural and functional relationships revealed by comparative analysis of syntenic genes in rhizobiales . BMC Evolutionary Biology , 5 ( 1 ). ↵ Hou , J. ( 2017 ). New approaches of protein function prediction from protein interaction networks . Academic Press . ↵ Huynen , M. et al. ( 2000 ). Predicting protein function by genomic context: Quantitative evaluation and qualitative inferences . Genome Research , 10 ( 8 ), 1204 – 1210 . OpenUrl Abstract / FREE Full Text ↵ Hwang , Y. et al. ( 2024 ). Genomic language model predicts protein co-regulation and function . Nature Communications , 15 ( 1 ). ↵ Jones , P. et al. ( 2014 ). Interproscan 5: genome-scale protein function classification . Bioinformatics , 30 ( 9 ), 1236 – 1240 . OpenUrl CrossRef PubMed Web of Science ↵ Junier , I. and Rivoire , O. ( 2013 ). Synteny in bacterial genomes: Inference, organization and evolution . arXiv: Genomics . ↵ Kemp , T. J. and Alcock , N. W. ( 2017 ). 100 years of x-ray crystallography . Science Progress , 100 ( 1 ), 25 – 44 . OpenUrl CrossRef PubMed ↵ Kingma , D. P. and Ba , J. ( 2014 ). Adam: A method for stochastic optimization . ↵ Konc , J. et al. ( 2013 ). Structure-based function prediction of uncharacterized protein using binding sites comparison . PLoS computational biology , 9 ( 11 ), e1003341 . OpenUrl CrossRef ↵ Kulmanov , M. and Hoehndorf , R. ( 2019 ). Deepgoplus: improved protein function prediction from sequence . Bioinformatics , 36 ( 2 ), 422 – 429 . OpenUrl CrossRef ↵ Kulmanov , M. et al. ( 2021 ). Deepgoweb: fast and accurate protein function prediction on the (semantic) web . Nucleic Acids Research , 49 ( W1 ), W140 – W146 . OpenUrl CrossRef PubMed ↵ Kulmanov , M. et al. ( 2023 ). Deepgo-se: Protein function prediction as approximate semantic entailment . ↵ Lai , B. and Xu , J. ( 2021 ). Accurate protein function prediction via graph attention networks with predicted structure information . Briefings in Bioinformatics , 23 ( 1 ). ↵ Li , Y. et al. ( 2022 ). Contrasting modes of macro and microsynteny evolution in a eukaryotic subphylum . Current Biology , 32 ( 24 ), 5335 – 5343 .e4. OpenUrl CrossRef PubMed ↵ Lin , Z. et al. ( 2023 ). Evolutionary-scale prediction of atomic-level protein structure with a language model . Science , 379 ( 6637 ), 1123 – 1130 . OpenUrl CrossRef PubMed ↵ Makarova , K. S. et al. ( 2011 ). Defense islands in bacterial and archaeal genomes and prediction of novel defense systems . Journal of Bacteriology , 193 ( 21 ), 6039 – 6056 . OpenUrl Abstract / FREE Full Text ↵ Markwick , P. R. L. et al. ( 2008 ). Structural biology by nmr: Structure, dynamics, and interactions . PLoS Computational Biology , 4 ( 9 ), e1000168 . OpenUrl CrossRef PubMed ↵ Mikolov , T. et al. ( 2013 ). Efficient estimation of word representations in vector space . ↵ Miller , D. et al. ( 2022 ). Deciphering microbial gene function using natural language processing . Nature Communications , 13 ( 1 ). ↵ Nguyen , C. D. et al. ( 2011 ). Protein annotation from protein interaction networks and gene ontology . Journal of biomedical informatics , 44 ( 5 ), 824 – 829 . OpenUrl CrossRef PubMed ↵ Osbourn , A. E. and Field , B. ( 2009 ). Operons . Cellular and Molecular Life Sciences , 66 ( 23 ), 3755 – 3775 . OpenUrl CrossRef PubMed Web of Science ↵ Parks , D. H. et al. ( 2017 ). Recovery of nearly 8, 000 metagenome-assembled genomes substantially expands the tree of life . Nature Microbiology , 2 ( 11 ), 1533 – 1542 . OpenUrl CrossRef PubMed ↵ Paszke , A. et al. ( 2019 ). Pytorch: An imperative style, high-performance deep learning library . ↵ Pedregosa , F. et al. ( 2011 ). Scikit-learn: Machine learning in Python . Journal of Machine Learning Research , 12 , 2825 – 2830 . OpenUrl ↵ Radivojac , P. et al. ( 2013 ). A large-scale evaluation of computational protein function prediction . Nature Methods , 10 ( 3 ), 221 – 227 . OpenUrl CrossRef PubMed ↵ Raiaan , M. A. K. et al. ( 2024 ). A review on large language models: Architectures, applications, taxonomies, open issues and challenges . IEEE Access , 12 , 26839 – 26874 . OpenUrl CrossRef ↵ Rogozin , I. B. ( 2002 ). Connected gene neighborhoods in prokaryotic genomes . Nucleic Acids Research , 30 ( 10 ), 2212 – 2223 . OpenUrl CrossRef PubMed Web of Science ↵ Simakov , O. et al. ( 2022 ). Deeply conserved synteny and the evolution of metazoan chromosomes . Science Advances , 8 ( 5 ). ↵ Steinegger , M. and Söding , J. ( 2017 ). Mmseqs2 enables sensitive protein sequence searching for the analysis of massive data sets . Nature Biotechnology , 35 ( 11 ), 1026 – 1028 . OpenUrl CrossRef PubMed ↵ Tukey , J. W. ( 1949 ). Comparing individual means in the analysis of variance . Biometrics , 5 ( 2 ), 99 – 114 . OpenUrl CrossRef PubMed Web of Science ↵ Wan , C. et al. ( 2019 ). Using deep maxout neural networks to improve the accuracy of function prediction from protein interaction networks . PLOS ONE , 14 ( 7 ), e0209958 . OpenUrl CrossRef PubMed ↵ Wolf , T. et al. ( 2019 ). Huggingface’s transformers: State-of-the-art natural language processing . ↵ Wolf , Y. I. et al. ( 2001 ). Genome alignment, evolution of prokaryotic genome organization, and prediction of gene function using genomic context . Genome Research , 11 ( 3 ), 356 – 372 . OpenUrl Abstract / FREE Full Text ↵ You , R. et al. ( 2021 ). Deepgraphgo: graph neural network for large-scale, multispecies protein function prediction . Bioinformatics , 37 ( Supplement_1 ), i262 – i271 . OpenUrl CrossRef PubMed View the discussion thread. Back to top Previous Next Posted October 17, 2024. Download PDF Supplementary Material Data/Code Email Thank you for your interest in spreading the word about bioRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following Context-based protein function prediction in bacterial genomes Message Subject (Your Name) has forwarded a page to you from bioRxiv Message Body (Your Name) thought you would like to see this page from the bioRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share Context-based protein function prediction in bacterial genomes Daulet Toibazar , Maxat Kulmanov , Robert Hoehndorf bioRxiv 2024.10.14.618363; doi: https://doi.org/10.1101/2024.10.14.618363 Share This Article: Copy Citation Tools Context-based protein function prediction in bacterial genomes Daulet Toibazar , Maxat Kulmanov , Robert Hoehndorf bioRxiv 2024.10.14.618363; doi: https://doi.org/10.1101/2024.10.14.618363 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Bioinformatics Subject Areas All Articles Animal Behavior and Cognition (7651) Biochemistry (17746) Bioengineering (13928) Bioinformatics (42066) Biophysics (21499) Cancer Biology (18650) Cell Biology (25579) Clinical Trials (138) Developmental Biology (13409) Ecology (19947) Epidemiology (2067) Evolutionary Biology (24374) Genetics (15633) Genomics (22557) Immunology (17775) Microbiology (40505) Molecular Biology (17217) Neuroscience (88796) Paleontology (667) Pathology (2845) Pharmacology and Toxicology (4836) Physiology (7664) Plant Biology (15179) Scientific Communication and Education (2047) Synthetic Biology (4304) Systems Biology (9839) Zoology (2272)

Text is read by the "Ask this paper" AI Q&A widget below. Extraction quality varies by source — PMC NXML preserves structure cleanly, OA-HTML may include some navigation residue, and OA-PDF can have broken hyphenation. The publisher copy (via DOI) is the canonical version.

My notes (saved in your browser only)

Ask this paper AI returns verbatim quotes from the full text · source: preprint-html

Answers must be backed by verbatim quotes from this paper's full text. Hallucinated quotes are dropped automatically; if no verbatim passage answers the question, we say so. How this works

Citation neighborhood (no data yet)

We don't have any in-corpus citations linked to this paper yet. This is a recent paper (2024) — citers typically take a year or two to land, and the OpenAlex reference graph may still be filling in.

Source provenance

europepmc
last seen: 2026-05-20T01:45:00.602351+00:00