Full text
52,066 characters
· extracted from
preprint-html
· click to expand
A curriculum learning approach to training antibody language models | bioRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-M677548'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search New Results A curriculum learning approach to training antibody language models View ORCID Profile Sarah M. Burbach , View ORCID Profile Bryan Briney doi: https://doi.org/10.1101/2025.02.27.640641 Sarah M. Burbach 1 Department of Immunology and Microbiology, The Scripps Research Institute , La Jolla, CA 92037 USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Sarah M. Burbach Bryan Briney 1 Department of Immunology and Microbiology, The Scripps Research Institute , La Jolla, CA 92037 USA 2 Center for Viral Systems Biology, The Scripps Research Institute , La Jolla, CA 92037 USA 3 Multi-omics Vaccine Evaluation Consortium, The Scripps Research Institute , La Jolla, CA 92037 USA 4 Scripps Consortium for HIV/AIDS Vaccine Development, The Scripps Research Institute , La Jolla, CA 92037 USA 5 San Diego Center for AIDS Research, The Scripps Research Institute , La Jolla, CA 92037 USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Bryan Briney For correspondence: briney{at}scripps.edu Abstract Full Text Info/History Metrics Data/Code Preview PDF ABSTRACT There is growing interest in pre-training antibody language models ( AbLMs ) with a mixture of unpaired and natively paired sequences, seeking to combine the proven benefits of training with natively paired sequences with the massive scale of unpaired antibody sequence datasets. However, given the novelty of this strategy, the field lacks a systematic evaluation of data processing methods and training strategies that maximize the benefits of mixed training data while accommodating the significant imbalance in the size of existing paired and unpaired datasets. Here we introduce a method of curriculum learning for AbLMs, which facilitates a gradual transition from unpaired to paired sequences during training. We optimize this method and show that a 650M-parameter curriculum model, CurrAb, outperforms existing mixed AbLMs in downstream classification tasks. INTRODUCTION Antibodies are a diverse and essential component of the adaptive immune system, with total available repertoire diversity estimated as high as 10 18 unique antibodies. 1 This exceptional diversity results initially from the somatic recombination of germline gene segments and is further refined upon antigen exposure via clonal expansion, somatic hypermutation ( SHM ), and antigen-driven selection of productive mutations. Within each recombined antibody gene, diversity is greatest in the complementary determining regions ( CDRs ), which comprise the antigen-recognition site and thus determine antibody specificity. Given the enormous diversity of the antibody repertoire, antibody language models ( AbLMs ) have grown in popularity for their potential to speed up the discovery and engineering of novel antibodies, particularly due to their ability to learn immunological mechanisms such as affinity maturation. 2 Because antibody recombination is a modular process, AbLMs tend to learn germline-encoded features easily but struggle with mutated residues and in the primarily non-templated CDR3. 3 , 4 Given this fact, a major focus of current studies is to improve model performance in the CDR regions. We have recently shown that pre-training an AbLM with natively paired antibody sequences rather than unpaired sequences improves the model’s ability to learn immunologically significant features that span both heavy and light chains, including heightened cross-chain attention on the functionally critical CDRs. 4 Despite the proven training benefits of paired sequences, available paired antibody sequence datasets are much smaller than unpaired datasets, and it is well documented that the cross-entropy ( CE ) loss of language models ( LMs ) scales with dataset size. 5 , 6 Given this, recent models 3 , 7 , 8 have incorporated a mix of unpaired and natively paired sequences, with the intuition that we can supplement the limited paired data with the larger quantities of unpaired data available. In theory, this should also improve a model’s ability to learn non-germline residues, since it sees a more diverse collection of somatically mutated antibodies in the larger unpaired datasets. Models trained with a mixture of unpaired and paired data typically pre-train using unpaired sequences and fine-tune with paired sequences. 3 IgBert and IgT5 from Kenlay et al 2024 7 introduced a variation of this approach, using a 2:1 ratio of unpaired to paired data during the fine-tuning phase to help mitigate catastrophic forgetting. 9 However, due to the recency of this pre-training shift, we lack a systematic evaluation of pretraining strategies to determine the optimal parameters for a model trained on a mixture of these datatypes. We theorized that a smoother transition between unpaired and paired sequences may help mitigate the risk of catastrophic forgetting further. Specifically, we introduce a method of training AbLMs inspired by curriculum learning, 10 which has been successful in the natural language processing domain. Curriculum learning involves ordering the training data, traditionally from ‘easiest’ to ‘hardest’. One such method is length-based curriculum learning, where training begins with shorter examples and transitions to longer ones. 11 To apply this to AbLMs, we ordered the data to start with the less information-dense unpaired sequences and then transition to the more information-dense paired sequences gradually during training. We hypothesized that this may enable the model to balance its knowledge of both sequence types without catastrophic forgetting while still emphasizing the paired sequences. We will compare this curriculum approach with other training methods including single data-type models, the traditional fine-tuning approach, and a constant approach (where data is mixed throughout training). In addition, we will train a large 650M-parameter curriculum model, CurrAb, and compare its performance on downstream tasks to existing mixed AbLMs. RESULTS Implementing curriculum learning for antibodies Given the unique challenges associated with pre-training on both unpaired and paired sequence data, we introduce a modified version of curriculum learning for AbLMs. This method modifies the way training examples are sampled, such that a variable percentage of unpaired sequences is selected based on the training step. The highest percentage of unpaired sequences are selected at the beginning of training and this percentage decreases as training progresses, to emphasize paired sequences in the final stages of training. To implement this, we generate an unpaired probability curve that determines what fraction of each training batch should be sampled from unpaired and paired datasets at each step of training ( Fig 1 ). The probability curve is a sigmoid decay function that enables the gradual transition from unpaired to paired sequences that can be modified to match training needs. Download figure Open in new tab Figure 1. Curriculum learning implementation for AbLMs. The approach is designed to help handle the massive data imbalance between unpaired and paired datasets. To implement this, an unpaired probability curve determines the percentage of data sampled from the unpaired dataset for a given batch at a given step. This percentage decreases as training progresses, to allow for an emphasis on paired sequences near the end of training. To optimize the curriculum strategy, we tested a variety of hyperparameters general to mixed-data models as well as curriculum-specific parameters, such as modifications to the unpaired probability curve and the learning rate schedule. To efficiently evaluate a large number of pre-training parameters, we initially used a pilot-scale 55M parameter variant of the ESM-2 12 architecture, with a slightly modified vocab of 33 characters that includes a unique token. These models were trained for 100k steps on a subset (20%) of the full paired and unpaired datasets. Optimizing model architecture and data processing for mixed-data models Prior to implementing the curriculum learning approach, we assessed the impact of a few key architecture and data formatting parameters for mixed-data models generally. Most significant are the changes observed based on the positional embedding ( PE ) type. Rotary position embeddings ( RoPE ) 13 consistently allow LMs to converge more quickly and reach a lower final loss than absolute PEs. By rotating the keys and queries based on their relative distance, RoPE can better accommodate sequences of varying lengths. Given that paired antibody sequences are approximately twice the length of unpaired ones, we hypothesized that RoPE would be especially impactful for AbLMs trained with a mixture of paired and unpaired sequences. To test this, we trained four models: two trained using only paired sequences (paired-only models) and two trained using only unpaired sequences (unpaired-only models), with either RoPE or absolute PE. Each of these models was then evaluated using both paired and unpaired test datasets. As expected, RoPE models perform slightly better on cross-entropy ( CE ) loss than absolute PE models when tested with the same data type they were trained on (paired-only models tested with paired sequences and unpaired-only models tested with unpaired sequences). Notably, when evaluated using heterologous data (paired test sequences for unpaired-only models or unpaired test sequences for paired-only models), RoPE models outperform absolute PE models by a large margin ( Fig 2A, 2C ). In the paired-only models, the absolute PE model performs similarly to the RoPE model in tests containing unpaired heavy chains, but significantly worse across all regions in unpaired light chains ( Fig 2B ). Similarly, in the unpaired-only models, the performance of the absolute PE model on paired sequences gets progressively worse in the heavy chain and is significantly worse in the light chain, compared to the RoPE model ( Fig 2D ). Our observation that the majority of the difference in performance between absolute PE and RoPE models is observed in the light chains, may be explained by their placement after heavy chains in paired sequence inputs. Download figure Open in new tab Figure 2. Comparing RoPE and absolute PE with paired and unpaired models. Two (A-B) paired-only model and (C-D) two unpaired-only models were trained with either RoPE or absolute PE. (A, C) CE loss on paired and unpaired test datasets of ∼10k sequences each. (B) Per-position CE loss of the paired-only models on 1k sequences from the unpaired test dataset. (C) Per-position CE loss of the unpaired-only models on 1k sequences from the paired test dataset. We next tested different separator token(s) and varying ratios of unpaired:paired data in mixed-data models. First, we note that a single separator token produced slight but consistent improvement on both CE loss and classifier performance, regardless of whether the separator token was uniquely used for chain separation () or a reuse of the start-of-string token () ( Table S1 ). Interestingly, adding a separator to unpaired sequences (immediately following heavy chains or immediately preceding light chains) resulted in better performance than omitting the separator, suggesting that a separator is useful even in an unpaired-only model. Second, we observe that increasing training time for one data type consistently results in a lower loss of that data type, but it always comes at the cost of increased loss on the other data type. Once the total training dataset exceeds 50% paired sequences, the paired loss fails to improve any further ( Table S2 ). Based on this, we train all subsequent mixed-data models with 62.5% unpaired data (just over 1 epoch of our total unpaired training dataset). Optimizing curriculum learning for antibodies To optimize our curriculum learning method, we tested a variety of modifications to the unpaired probability curve and learning rate ( LR ) schedule. We first tested different ranges of probabilities for these curves, with the widest ranging from 1.0 to 0.0 (max1) and the narrowest ranging from 0.7 to 0.3 (max0.7) ( Fig 3A ). Performance on the unpaired test set reveals that the CE loss decreases as the probability range narrows ( Fig 3B ). Performance on the paired test set is much less variable, with the max0.8 range resulting in the lowest loss by a small margin. These results suggest that including a mix of data types (with a minimum of ∼20% of the minority data type) throughout training is useful to the models’ final performance. Second, we tested the impact of the slope of the unpaired probability curve decay, which is determined by the k-value. Models with a larger k-value decay at a steeper rate than models with a smaller k-value, resulting in a more rapid transition from unpaired to paired training data ( Fig 3C ). Changing the slope has almost no effect on the paired performance, with k = 50 performing only slightly better than the others ( Fig 3D ). We observe a slightly larger impact on the unpaired loss, with k = 15 resulting in the lowest CE loss. Download figure Open in new tab Figure 3. Optimizing hyperparameters for curriculum models. (A-B) Testing different ranges of the unpaired probability curve. (C-D) Testing different slopes of the unpaired probability curve, by adjusting their k-value. (E-F) Testing three different LR schedules: linear, WSD, and SGDR. All models were evaluated based on their CE loss on the paired and unpaired test datasets. Finally, we tested different learning rate schedules to determine if a higher learning rate later in training is useful for learning paired sequences. We evaluated a linear-decay schedule, a warmup-stable-decay ( WSD ) 14 schedule, and a cosine annealing schedule (commonly known as stochastic gradient descent with warm restarts ( SGDR )) 15 ( Fig 3E ). We find that the linear LR results in the lowest unpaired test loss, while WSD results in the lowest test loss for paired sequences ( Fig 3F ). The improved performance of the WSD LR suggests that a higher learning rate during the transition can assist the model in learning paired sequences. However, this small gain in paired performance comes at a much larger cost to unpaired performance. Based on the limited improvements observed in the paired CE loss, we optimized the curriculum implementation based on the unpaired performance. Therefore, our optimized parameters are a probability range of 0.7 to 0.3, a k-value of 15, and a linear LR schedule. Assessing methods for training mixed models To directly compare strategies for training with mixed datasets, we trained a series of five models: one model pre-trained on unpaired data and finetuned on paired data ( finetuned ); one model trained with a constant ratio of paired and unpaired data ( constant ), one model trained using our curriculum-training strategy that dynamically adjusts the ratio of paired and unpaired data throughout training ( curriculum ), and two control models trained using only unpaired ( unpaired-only ) or paired ( paired-only ) data ( Fig 4A ). Each model was otherwise identical in architecture and was trained for 100k steps. The 50k checkpoint was used for downstream testing of the paired-only model due to overfitting. The paired:unpaired ratio of the constant model and finetuning duration of the finetuned model were selected to ensure that all mixed models were trained with the same total amount (in epochs) of unpaired and paired training data. Download figure Open in new tab Figure 4. Comparing the performance of mixed model training methods. (A) Unpaired probability curves for the five models. (B) CE loss on paired and unpaired test datasets of ∼10k sequences. (C) Mixed models accuracy at predicting CDRH3 of 1k paired sequences from the test set. Results for the three classification tasks, which are one (D) Native vs Random pair classification and two Healthy Donor vs Flu vs CoV specificity classifications using (E) paired and (F) unpaired sequences. (G) The results of the pair classification were additionally split by mutated pairs, unmutated pairs, and mismatched pairs. Metrics on classification tasks are mean and standard error, with the highest values bolded and the second highest values underlined. We evaluated each model with a masked language modeling objective using the held-out test data ( Fig 4B ) and observed that the curriculum and constant models performed best across CE loss and accuracy on the paired test set. We additionally observed that the unpaired-only model performed best across both CE loss and accuracy on the unpaired test set, followed by the curriculum model. The boost in unpaired performance in the curriculum model compared to the other mixed models suggests that the curriculum training schedule enables the model to retain its knowledge of unpaired sequences the best. To further assess the paired accuracy of the mixed models, we also calculated the CDRH3 accuracy for these three models ( Fig 4C ). We observed that the CDRH3 accuracy is essentially the same across all mixed models, indicating that the method of training a mixed model does not have a significant impact on the model’s understanding of the CDRH3 region. To further assess model performance, we performed a pair classification task, introduced in Ng and Briney 2024 16 . The goal of this binary classification task is to identify whether a given paired sequence is a native or random pairing. The dataset was generated from the paired test dataset and is composed of naive (unmutated) and memory (mutated) sequences. The paired-only model performed best, followed by the curriculum model which outperforms the other mixed data models ( Fig 4D ). Next, we separated the test data for pair classification into three subsets: pairs in which both chains are mutated ( mutated ), pairs in which both chains are unmutated ( unmutated ), or pairs in which one chain is mutated while the other is not ( different ) ( Fig 4G ). All models except the unpaired model perform significantly better on the different subset, indicating that pair classification is driven, at least in par,t by identifying similar levels of mutation across both chains. Performance on the mutated subset is most varied between models, with the paired-only and curriculum models showing the highest classification accuracy. We next performed three-way specificity classification tasks, in which the model was finetuned to classify between coronavirus ( CoV ) specific antibodies, influenza ( Flu ) specific antibodies, and nonspecific healthy donor ( HD ) antibodies. Models were fine-tuned with either paired sequences ( Fig 4E ) or unpaired heavy chain sequences ( Fig 4F ). The paired-only model outperformed all models across both the paired and unpaired tasks, followed closely by the finetuned model. Results for a two-way specificity classification task (CoV vs HD) can be found in Table S3 , where we observe similar results but exhibit smaller margins of difference. This suggests that a strong emphasis on paired sequences (despite the previously observed forgetting of unpaired sequences) is of primary importance for specificity classification tasks. The higher performance of the paired-only model compared to the mixed models seems to be an artifact of the small size of our pilot-scale models, however. When the same classification tasks are performed with larger 650M-parameter models, the curriculum model outperforms the paired-only model ( Table S4 ) across all tasks. Large-scale curriculum model outperforms existing mixed models A distinct advantage of mixed models over paired-only models is the massively increased training data scale, which could allow longer training of larger models before overfitting. To test the scalability of our curriculum method, we trained CurrAb: a 650M-parameter model based on the ESM-2 architecture. The model was trained for 500k steps using the optimized curriculum schedule described above and a reduced learning rate (1e-4) to stabilize training. We compared the performance of CurrAb to several existing mixed models: IgBERT 7 , IgT5 7 , AbLang2 3 , and AntiBERTa2 8 , using the pair classification task described previously. CurrAb performed better than all other mixed models across all metrics except F1, followed most closely by AbLang2 ( Fig 5A ). The unexpectedly low performance of IgBERT and AntiBERTa2 may be due to the classification head of BERT models, which is only a single projection layer that may not be sufficient for this complex task. We additionally performed the three-way specificity classification with CoV-specific, Flu-specific, and HD sequences. CurrAb again outperformed other mixed models on both the paired ( Fig 5B ) and unpaired ( Fig 5C ) specificity classification tasks. The second-best model performances are AbLang2 in paired specificity classification and AntiBERTa2 in unpaired specificity classification. A closer examination of the paired specificity classification results showed that CurrAb was >10% more accurate than the next best-performing model, AbLang2 ( Fig 5D ). Download figure Open in new tab Figure 5. Comparing CurrAb to existing mixed data models. Results of three classification tasks, which are one (A) Native vs Random pair classification and two Healthy Donor vs Flu vs CoV specificity classification tasks using (B) paired and (C) unpaired sequences. (D) The bar plot shows the mean accuracy and standard error of the pair classification task. Metrics on classification tasks are mean and standard error, with the highest values bolded and the second highest values underlined. IgBERT and IgT5 were trained with an approach closest to the curriculum strategy – these models were pre-trained using unpaired sequences and finetuned with a mix of unpaired and paired sequences. Despite this similar training approach and the large size of IgT5 (3B parameters), these models tend to perform poorly on specificity classification tasks. This suggests that model architecture (particularly rotary embeddings) and a larger emphasis on paired sequences is critical to high performance on these tasks. DISCUSSION Recent studies on AbLMs have shown that the benefits of training on natively paired sequences can be sufficient to overcome a significant disadvantage in training data scale compared to unpaired sequences. The high cost and effort required to recover natively paired antibody sequences has sparked interest in methods that maximize the training value of these limited datasets, including supplementing paired sequences with unpaired data. Theoretically, this mixed training approach should still allow the model to learn critical cross-chain features while leveraging the greater diversity in much larger unpaired sequence datasets. However, it is not clear how best to combine these two data types to maximize training value. We first performed a systematic analysis of model architectures and data formatting methods to understand their effect on mixed model training. The most impactful change was the use of rotary embeddings, which delivered a surprisingly large improvement in mixed-data model performance. Compared to absolute PE, RoPE greatly reduced the loss of homogeneous models (paired-only or unpaired-only) when tested with the heterologous data type. This is strong evidence that mixed-data models should use rotary embeddings whenever possible. In addition, we noted that mixed-data models using a single chain separator token (whether that be or ) performed best, with the token showing a slight performance benefit in homogeneous models as well. Next, we introduced a curriculum learning approach to training AbLMs, which allows for a more gradual transition from unpaired to paired sequences. Including both data types throughout the entire training course, with an increased emphasis on paired sequences toward the end of training, resulted in the best final model performance. We also showed that adjusting the parameters of the unpaired probability curve can differentially affect model performance on unpaired or paired sequences, meaning the curriculum approach can be tuned for specific performance objectives. Generally, optimizing for unpaired loss resulted in larger overall performance gains on downstream tasks, perhaps because unpaired sequences comprise the bulk of the training data. We expected that one of the primary performance-enhancing benefits of supplementing paired training data with unpaired sequences would be the ability to train larger models for longer without overfitting. CurrAb, a 650M parameter curriculum model, validated this assumption by outperforming existing models on various classification tasks including native pairing and antigen specificity. The magnitude of improvement over IgBERT and IgT5 was surprising, particularly given that these models were finetuned with a mixture of paired and unpaired sequences. While this strategy is similar to our curriculum training approach in that the final training stages do not focus exclusively on a single data type, we note two differences that may underlie this performance disparity. First, it is possible that exposure to paired sequences only during finetuning is insufficient for the model to learn cross-chain features fully; instead, paired sequences may need to be introduced at the earliest stages of training to capture their full training value. Second, architectural optimizations such as rotary position embeddings, which are lacking in IgBERT and IgT5, may yield substantial performance improvements on downstream classification tasks, particularly those that involve paired antibody sequences. More broadly, these findings emphasize the importance of generating larger datasets of paired antibody sequences, as paired-only models approach the performance of mixed models trained with substantially more data. In other words, the marginal training value of new data appears to strongly favor paired antibody sequences, despite the high cost of generating natively paired antibody sequences. It may also be useful to apply additional training methods that focus learning on the CDR regions, such as preferential masking 16 and focal loss 3 , 17 , to use both the paired and unpaired data more efficiently. Beyond AbLMs, the ability of curriculum-based training approaches to prevent catastrophic forgetting is likely to be beneficial for a variety of biological models for which cross-domain expertise is important. For example, a curriculum model trained with a combination of proteins and antibodies could perform as well as a specialized antibody sequence or structure model while retaining an expert-level understanding of general proteins. Such a model could be very well-suited for tasks like antibody-antigen docking. In summary, we report three important findings. First, RoPE significantly impacts the ability of AbLMs to generalize across unpaired and paired data, highlighting positional embeddings as a key but underappreciated component of mixed-data biological language models. Second, curriculum AbLMs typically perform best when optimized for unpaired performance, but the most important factor for final model performance is ensuring a sufficient emphasis on paired sequences throughout the entire training course. Finally, curriculum learning effectively mitigates catastrophic forgetting, making it a useful strategy for training biological models that exhibit uniformly high performance across multiple domains. METHODS Training Data Sequence data was downloaded from the OAS 18 on September 12th, 2024. In addition to sequences from the OAS, the paired dataset was supplemented with an internally generated dataset of ∼400k paired sequences. Filtering and clustering were performed as described in AntiRef 19 , with additional filtering for sequences containing ‘nan’ characters. The dataset clustered at 90% was chosen for both datasets, resulting in 151,764,423 unpaired sequences and 1,717,423 paired sequences in the full datasets. The final 650M-parameter model, CurrAb, was trained on the full dataset and used 96% of the data for training, with the remaining 4% left out for the evaluation and test sets. For compute efficiency, initial tests in Figures 2 - 4 were trained with a downsampled dataset (20% of the full dataset), resulting in 30,352,885 unpaired sequences and 343,485 paired sequences. These datasets were similarly split to use 96% of the sequences for training and the remaining 4% for the evaluation and test sets. Test sets used for CE loss in Figures 2 , 3 and 4 were the paired test set and a sample of ∼10k sequences from the unpaired test set (to match the number of sequences in the paired test set). For the specificity classification tasks, two datasets were generated. CoV-specific sequences were downloaded from CoV-AbDab 20 on November 11th, 2024 and an internal donor L1236. Flu-specific sequences were obtained from Wang et al. 21 and filtered for paired sequences only. Healthy donor sequences were obtained from memory B-cell sequences from the Jaffe et al. 22 and Phad et al. 23 datasets and filtered to exclude any sequences in the train and evaluation datasets to prevent data leakage for both model sizes. For the HD-Flu-CoV classifications, the datasets were clustered by class at 99%, then combined with an equal number of sequences in each class, resulting in a final dataset with 4,398 sequences. For the HD-CoV classifications, the datasets were clustered by class at 95%, then combined with an equal number of sequences in each class, resulting in a final dataset with 27,442 sequences. Both classification datasets were split for 5-fold cross-validation with stratification. The same datasets were used for the unpaired versions of these tasks, but only the heavy chain sequence was provided. For the pair classification task, the paired sequences from the full-data test set were filtered to exclude any sequences in the downscaled train and evaluation datasets, to prevent data leakage for both model sizes. This resulted in 41,564 paired sequences, which are comprised of both naive and memory B-cell sequences. The dataset was processed as described in Ng and Briney 2024 16 , including a split for 5-fold cross-validation with stratification. Curriculum Implementation To enable training of different types of mixed models (constant and curriculum) modifications were made to the HuggingFace 24 trainer using a trainer callback and by subclassing the PyTorch 25 dataset class. This facilitated the separation of unpaired and paired datasets (for logging and evaluation purposes) and allowed for tracking and updating the unpaired probability function throughout training. The unpaired probability function determined what percentage of sequences should be sampled from the unpaired dataset at each step during training. The unpaired probability P ( t ) for the curriculum models is represented with the following equation: where t equals the current step divided by the total train steps. The values of A (height of the curve) and B (the vertical shift, aka upper bound of the curve) were modified to adjust the range of the probabilities, the k value was modified to adjust the slope of the sigmoid function, and the shift values were calculated to ensure that the total unpaired percentage (ie. 62.5%) was achieved. Refer to Table S6 for the specific values used for each curriculum model. Model Pre-Training All models used a slightly modified ESM-2 architecture model. This is an encoder-only architecture that produces embeddings useful for downstream tasks such as specificity classification, pair classification, and structure prediction. Models were trained with an MLM objective. This means that 15% of the input sequence was selected for prediction and of these, 80% were replaced with a token, 10% were replaced with a random token from the vocabulary, and 10% were left unchanged. All models used a slightly modified ESM-2 vocabulary of 33 tokens, with an added token. Based on the separator tests presented in Table 1 , the large-scale models were trained with a separator. Sequences were preprocessed to place the separator between the heavy and light chains of the paired sequences, at the end of unpaired heavy chains, and at the beginning of unpaired light chains. Inputs were padded to 320 tokens, to accommodate the longest paired sequence. For initial tests, we used a modified 55M-parameter ESM-2 architecture model, with 5 layers, 20 attention heads per layer, a hidden size of 960, and an intermediate size of 3840. After initial embedding tests in Fig 2 , RoPE was used for all subsequent models. Models were trained for 100k steps with a total batch size of 512. On 4 L40S graphics processing units (GPUs) using DeepSpeed ZeRO Stage 1 26 via the Accelerate 27 library, this equates to ∼11 hours per model. The peak learning rate was 4e-4, with a linear warmup for the first 6,000 steps followed by a linear decay. The only exception to this was the LR tests in Fig 3E-F , which tested WSD and SDGR LR schedules. The ideal parameters for WSD and SGDR were chosen based on Hägele et al 2024. 28 The final curriculum model used the 650M parameter ESM-2 architecture, with RoPE and used the unique token as the separator token in the paired and unpaired sequences. The curriculum parameters were chosen to balance unpaired and paired performance – the max0.7 unpaired probability schedule was selected to benefit unpaired performance while k=20 was chosen to benefit paired performance. The model was trained for 500k steps with a total batch size of 512, equating to ∼6 days on 8 A100 GPUs using DeepSpeed ZeRO Stage 1. The peak learning rate was reduced to 1e-4 to assist with model convergence, with a linear warmup for the first 30,000 steps followed by a linear decay. Models were logged using Weights & Biases (wandb). Data from wandb was used to produce the unpaired probability, LR schedule, and evaluation loss curves. Model Evaluation & Testing CE loss and accuracy calculations on the eval and test sets were performed using the HuggingFace trainers evaluate function, with a custom compute_metrics function provided to ensure the metrics exclude the separator tokens from calculations. Accuracy was calculated using scikit-learn and CE loss was calculated using PyTorch. CDRH3 accuracy was calculated by masking and predicting each position in the CDRH3 region individually, then calculating the accuracy of those predictions by averaging. Classification Tasks For all classification tasks, models were finetuned with a standard sequence classification head. For all models we trained and models hosted on HuggingFace (IgBERT, IgT5, and AntiBERTa2), the base models were loaded with the default sequence classification head for that model architecture. For AbLang2, which is only available as a pip package, the sequence classification head was added manually and modeled after the ESM-2 classification head. For specificity classification tasks, all models had a warmup ratio of 0.1 and a peak learning rate of 5e-5 followed by a linear decay. For binary classification (HD vs CoV), models were trained for 10 epochs and a total batch size of 32 per step. For 3-class classification (HD vs Flu vs CoV), models were trained for 5 epochs and a total batch size of 8 per step. For pair classification tasks, models were finetuned following the training schedule presented in Ng and Briney 2024 16 . Models were trained for 50 epochs, with a warmup ratio of 0.1, a peak learning rate of 1e-5, and a total batch size of 256 per step. Metrics used to evaluate the classifiers were: accuracy, F1, area under the receiver operating characteristic curve ( AUC ), area under the precision-recall curve ( AUPR ), and Matthews correlation coefficient ( MCC ). Metrics on the test set were averaged across cross-validation runs and standard error was calculated. Scikit-learn was used to calculate these metrics. Code and Data Availability The code used for model training, evaluation, and figure generation is available on GitHub ( github.com/brineylab/curriculum-paper ). The training data and model weights for CurrAb are available on Zenodo ( doi.org/10.5281/zenodo.14661302 ) and CurrAb is also available to use through Hugging Face ( huggingface.co/brineylab/CurrAb ). AUTHOR CONTRIBUTIONS Conceptualization: SB, BB Model training and evaluation: SB Manuscript preparation and revisions: SB, BB FUNDING This work was funded by the National Institutes of Health (P01-AI177683, U19-AI135995, R01-AI171438, P30-AI036214, and UM1-AI144462) and the Pendleton Foundation. DECLARATION OF INTERESTS BB is an equity shareholder in Infinimmune and a member of their Scientific Advisory Board. SUPPLEMENTARY INFORMATION View this table: View inline View popup Download powerpoint Table S1. CE Loss and accuracy for separator token tests. Mixed, paired-only, and unpaired-only models were trained with 5 different separators. The separators tested were: no separator, , , , and where is the BOS token and is a unique separator token. Separators are placed between chains in paired sequences and unpaired sequences based on the chain (end of the heavy chains and the beginning of the light chains). Models were assessed on paired and unpaired test datasets, each containing ∼10k sequences. View this table: View inline View popup Download powerpoint Table S2. CE loss and accuracy for different unpaired percentages. Mixed models were trained with increasing percentages of unpaired data. Models were assessed on paired and unpaired test datasets, each containing ∼10k sequences. View this table: View inline View popup Download powerpoint Table S3. Additional classification tasks to compare mixed model training methods. Results for ‘HD vs CoV’ specificity classification task with paired and unpaired sequences. Metrics on classification tasks are mean and standard error, with the highest values bolded and the second highest values underlined View this table: View inline View popup Download powerpoint Table S4. Classification tasks on 650M-parameter paired-only and curriculum models. Results specificity classification tasks (HD vs CoV’ and ‘HD vs Flu vs CoV’) with paired and unpaired sequences and the pair classification task. Metrics on classification tasks are mean and standard error, with the highest values bolded. View this table: View inline View popup Download powerpoint Table S5. Additional classification tasks to compare large-scale models. Results for ‘HD vs CoV’ specificity classification task with paired and unpaired sequences. Metrics on classification tasks are mean and standard error, with the highest values bolded and the second highest values underlined. View this table: View inline View popup Download powerpoint Table S6. Unpaired probability equation values for curriculum models. Corresponding figure, model, and equation values (A, B, shift, and k) are listed for each model. Italics represent the values being tested in the given figure. Footnotes https://github.com/brineylab/curriculum-paper https://huggingface.co/brineylab/CurrAb https://doi.org/10.5281/zenodo.14661302 REFERENCES 1. ↵ Briney , B. , Inderbitzin , A. , Joyce , C. , and Burton , D.R. ( 2019 ). Commonality despite exceptional diversity in the baseline human antibody repertoire . Nature 566 , 393 – 397 . doi: 10.1038/s41586-019-0879-y . OpenUrl CrossRef PubMed 2. ↵ Ruffolo , J.A. , Gray , J.J. , and Sulam , J. ( 2021 ). Deciphering antibody affinity maturation with language models and weakly supervised learning . arXiv [q-bio.BM] . 3. ↵ Olsen , T.H. , Moal , I.H. , and Deane , C.M. ( 2024 ). Addressing the antibody germline bias and its effect on language models for improved antibody design . bioRxiv , 2024.02.02.578678. doi: 10.1101/2024.02.02.578678 . OpenUrl Abstract / FREE Full Text 4. ↵ Burbach , S.M. , and Briney , B. ( 2024 ). Improving antibody language models with native pairing . PATTER 0 . doi: 10.1016/j.patter.2024.100967 . OpenUrl CrossRef PubMed 5. ↵ Kaplan , J. , McCandlish , S. , Henighan , T. , Brown , T.B. , Chess , B. , Child , R. , Gray , S. , Radford , A. , Wu , J. , and Amodei , D. ( 2020 ). Scaling laws for neural language models . arXiv [cs.LG] . 6. ↵ Hoffmann , J. , Borgeaud , S. , Mensch , A. , Buchatskaya , E. , Cai , T. , Rutherford , E. , Casas , D. de L. , Hendricks , L.A. , Welbl , J. , Clark , A. , et al. ( 2022 ). Training Compute-Optimal Large Language Models . arXiv [cs.CL] . 7. ↵ Kenlay , H. , Dreyer , F.A. , Kovaltsuk , A. , Miketa , D. , Pires , D. , and Deane , C.M. ( 2024 ). Large scale paired antibody language models . arXiv [q-bio.BM] . 8. ↵ Barton , J. , Galson , J.D. , and Leem , J. ( 2024 ). Enhancing antibody language models with structural information . bioRxiv . doi: 10.1101/2023.12.12.569610 . OpenUrl Abstract / FREE Full Text 9. ↵ Kirkpatrick , J. , Pascanu , R. , Rabinowitz , N. , Veness , J. , Desjardins , G. , Rusu , A.A. , Milan , K. , Quan , J. , Ramalho , T. , Grabska-Barwinska , A. , et al. ( 2017 ). Overcoming catastrophic forgetting in neural networks . Proc. Natl. Acad. Sci. U. S. A . 114 , 3521 – 3526 . doi: 10.1073/pnas.1611835114 . OpenUrl Abstract / FREE Full Text 10. ↵ Bengio , Y. , Louradour , J. , Collobert , R. , and Weston , J. ( 2009 ). Curriculum learning . In Proceedings of the 26th Annual International Conference on Machine Learning (ACM) , p. 6 . doi: 10.1145/1553374.1553380 . OpenUrl CrossRef 11. ↵ Nagatsuka , K. , Broni-Bediako , C. , and Atsumi , M. ( 2022 ). Length-based curriculum learning for efficient pre-training of language models . New Gener. Comput . 41 , 109 – 134 . doi: 10.1007/s00354-022-00198-8 . OpenUrl CrossRef 12. ↵ Lin , Z. , Akin , H. , Rao , R. , Hie , B. , Zhu , Z. , Lu , W. , Smetanin , N. , Verkuil , R. , Kabeli , O. , Shmueli , Y. , et al. ( 2023 ). Evolutionary-scale prediction of atomic-level protein structure with a language model . Science 379 , 1123 – 1130 . doi: 10.1126/science.ade2574 . OpenUrl CrossRef PubMed 13. ↵ Su , J. , Lu , Y. , Pan , S. , Murtadha , A. , Wen , B. , and Liu , Y. ( 2021 ). RoFormer: Enhanced transformer with Rotary Position Embedding . arXiv [cs.CL] . 14. ↵ Hu , S. , Tu , Y. , Han , X. , He , C. , Cui , G. , Long , X. , Zheng , Z. , Fang , Y. , Huang , Y. , Zhao , W. , et al. ( 2024 ). MiniCPM: Unveiling the potential of Small Language Models with scalable training strategies . arXiv [cs.CL] . 15. ↵ Loshchilov , I. , and Hutter , F. ( 2016 ). SGDR: Stochastic gradient descent with warm restarts . arXiv [cs.LG] . 16. ↵ Ng , K. , and Briney , B. ( 2024 ). Focused learning by antibody language models using preferential masking of non-templated regions . Preprint , doi: 10.1101/2024.10.23.619908 10.1101/2024.10.23.619908. OpenUrl Abstract / FREE Full Text 17. ↵ Lin , T.-Y. , Goyal , P. , Girshick , R. , He , K. , and Dollár , P. ( 2017 ). Focal Loss for dense object detection . arXiv [cs.CV] . 18. ↵ Kovaltsuk , A. , Leem , J. , Kelm , S. , Snowden , J. , Deane , C.M. , and Krawczyk , K. ( 2018 ). Observed Antibody Space: A Resource for Data Mining Next-Generation Sequencing of Antibody Repertoires . J. Immunol . 201 , 2502 – 2509 . doi: 10.4049/jimmunol.1800708 . OpenUrl Abstract / FREE Full Text 19. ↵ Briney , B. ( 2022 ). AntiRef: reference clusters of human antibody sequences . bioRxiv , 2022.12.30.522338. doi: 10.1101/2022.12.30.522338 . OpenUrl Abstract / FREE Full Text 20. ↵ Raybould , M.I.J. , Kovaltsuk , A. , Marks , C. , and Deane , C.M. ( 2021 ). CoV-AbDab: the coronavirus antibody database . Bioinformatics 37 , 734 – 735 . doi: 10.1093/bioinformatics/btaa739 . OpenUrl CrossRef PubMed 21. ↵ Wang , Y. , Lv , H. , Lei , R. , Yeung , Y.-H. , Shen , I.R. , Choi , D. , Teo , Q.W. , Tan , T.J.C. , Gopal , A.B. , Chen , X. , et al. ( 2023 ). An explainable language model for antibody specificity prediction using curated influenza hemagglutinin antibodies . bioRxiv , 2023.09.11.557288. doi: 10.1101/2023.09.11.557288 . OpenUrl Abstract / FREE Full Text 22. ↵ Jaffe , D.B. , Shahi , P. , Adams , B.A. , Chrisman , A.M. , Finnegan , P.M. , Raman , N. , Royall , A.E. , Tsai , F. , Vollbrecht , T. , Reyes , D.S. , et al. ( 2022 ). Dataset: Functional antibodies exhibit light chain coherence . doi: 10.5281/zenodo.6348137 https://doi.org/10.5281/zenodo.6348137 . OpenUrl CrossRef 23. ↵ Phad , G.E. , Pinto , D. , Foglierini , M. , Akhmedov , M. , Rossi , R.L. , Malvicini , E. , Cassotta , A. , Fregni , C.S. , Bruno , L. , Sallusto , F. , et al. ( 2022 ). Clonal structure, stability and dynamics of human memory B cells and circulating plasmablasts . Nat. Immunol . 23 , 1076 – 1085 . doi: 10.1038/s41590-022-01230-1 . OpenUrl CrossRef PubMed 24. ↵ Wolf , T. , Debut , L. , Sanh , V. , Chaumond , J. , Delangue , C. , Moi , A. , Cistac , P. , Rault , T. , Louf , R. , Funtowicz , M. , et al. ( 2019 ). HuggingFace’s Transformers: State-of-the-art Natural Language Processing . arXiv [cs.CL] . 25. ↵ Ansel , J. , Yang , E. , He , H. , Gimelshein , N. , Jain , A. , Voznesensky , M. , Bao , B. , Bell , P. , Berard , D. , Burovski , E. , Chauhan , G. , Chourdia , A. , Constable , W. , Desmaison , A. , DeVito , Z. , Ellison , E. , Feng , W. , Gong , J. , Gschwind , M. , Hirsh , B. , Huang , S. , Kalambarkar , K. , Kirsch , L. , Lazos , M. , Lezcano , M. , Liang , Y. , Liang , J. , Lu , Y. , Luk , C. , Maher , B. , Pan , Y. , Puhrsch , C. , Reso , M. , Saroufim , M. , Siraichi , M. Y. , Suk , H. , Suo , M. , Tillet , P. , Wang , E. , Wang , X. , Wen , W. , Zhang , S. , Zhao , X. , Zhou , K. , Zou , R. , Mathews , A. , Chanan , G. , Wu , P. , & Chintala , S. ( 2024 ). PyTorch 2: Faster Machine Learning Through Dynamic Python Bytecode Transformation and Graph Compilation. In doi: 10.1145/3620665.3640366 . OpenUrl CrossRef 26. ↵ Rajbhandari , S. , Rasley , J. , Ruwase , O. , and He , Y. ( 2019 ). ZeRO: Memory Optimizations Toward Training Trillion Parameter Models . arXiv [cs.LG] . 27. ↵ Sylvain Gugge , Lysandre Debut , Thomas Wolf , Philipp Schmid , Zachary Mueller , Sourab Mangrulkar , Marc Sun , Benjamin Bossan ( 2022 ). Accelerate: Training and inference at scale made simple, efficient and adaptable (Github) . 28. ↵ Hägele , A. , Bakouch , E. , Kosson , A. , Allal , L.B. , Von Werra , L. , and Jaggi , M. ( 2024 ). Scaling laws and compute-optimal training beyond fixed training durations . arXiv [cs.LG] . View the discussion thread. Back to top Previous Next Posted March 02, 2025. Download PDF Data/Code Email Thank you for your interest in spreading the word about bioRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following A curriculum learning approach to training antibody language models Message Subject (Your Name) has forwarded a page to you from bioRxiv Message Body (Your Name) thought you would like to see this page from the bioRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share A curriculum learning approach to training antibody language models Sarah M. Burbach , Bryan Briney bioRxiv 2025.02.27.640641; doi: https://doi.org/10.1101/2025.02.27.640641 Share This Article: Copy Citation Tools A curriculum learning approach to training antibody language models Sarah M. Burbach , Bryan Briney bioRxiv 2025.02.27.640641; doi: https://doi.org/10.1101/2025.02.27.640641 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Immunology Subject Areas All Articles Animal Behavior and Cognition (7637) Biochemistry (17705) Bioengineering (13899) Bioinformatics (41970) Biophysics (21463) Cancer Biology (18605) Cell Biology (25526) Clinical Trials (138) Developmental Biology (13385) Ecology (19911) Epidemiology (2067) Evolutionary Biology (24329) Genetics (15615) Genomics (22514) Immunology (17743) Microbiology (40424) Molecular Biology (17194) Neuroscience (88650) Paleontology (667) Pathology (2835) Pharmacology and Toxicology (4827) Physiology (7648) Plant Biology (15160) Scientific Communication and Education (2046) Synthetic Biology (4302) Systems Biology (9825) Zoology (2271)
Text is read by the "Ask this paper" AI Q&A widget below.
Extraction quality varies by source — PMC NXML preserves structure
cleanly, OA-HTML may include some navigation residue, and OA-PDF can
have broken hyphenation. The publisher copy
(via DOI)
is the canonical version.