LMI4Boltz: Optimising VRAM utilisation to predict large macromolecular complexes with consumer grade hardware

preprint OA: closed CC-BY-4.0
📄 Open PDF Full text JSON View at publisher

Abstract

AlphaFold2 has revolutionised structural biology by enabling the prediction of protein structures approaching experimental quality. AlphaFold3 extends this framework to support modelling broad biomolecular classes while also reducing the computational cost of prediction. However, AlphaFold3 is distributed with licence conditions which restrict general purpose use. Boltz is a permissive, open-source re-implementation of AlphaFold3, but it is bottlenecked by increased VRAM requirements and requires high-end GPU hardware to model large molecular systems. Here we introduce Low Memory Inference for Boltz (LMI4Boltz) which reduces VRAM requirements using in-place updates, offloading tensors to host memory, careful management of functional scope and aggressive chunking of key operations. Using these strategies, LMI4Boltz increases the token size limit of Boltz-2 by 66.7% without sacrificing prediction accuracy. These optimisations improve the accessibility of Boltz using consumer grade hardware and unlock the ability to model large molecular systems. LMI4Boltz is available at https://github.com/tlitfin/lmi4boltz .
Full text 21,718 characters · extracted from preprint-html · click to expand
LMI4Boltz: Optimising VRAM utilisation to predict large macromolecular complexes with consumer grade hardware | bioRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-M677548'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search New Results LMI4Boltz: Optimising VRAM utilisation to predict large macromolecular complexes with consumer grade hardware View ORCID Profile Thomas Litfin , View ORCID Profile Joshua Storm Caley , View ORCID Profile Katharine A. Michie doi: https://doi.org/10.1101/2025.10.29.684571 Thomas Litfin 1 Structural Biology Facility, Mark Wainwright Analytical Centre, University of New South Wales , Sydney, NSW, Australia 2 Australian BioCommons, University of Melbourne , Melbourne, Victoria, Australia Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Thomas Litfin For correspondence: t.litfin{at}unsw.edu.au Joshua Storm Caley 1 Structural Biology Facility, Mark Wainwright Analytical Centre, University of New South Wales , Sydney, NSW, Australia Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Joshua Storm Caley Katharine A. Michie 1 Structural Biology Facility, Mark Wainwright Analytical Centre, University of New South Wales , Sydney, NSW, Australia Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Katharine A. Michie Abstract Full Text Info/History Metrics Supplementary material Preview PDF Abstract AlphaFold2 has revolutionised structural biology by enabling the prediction of protein structures approaching experimental quality. AlphaFold3 extends this framework to support modelling broad biomolecular classes while also reducing the computational cost of prediction. However, AlphaFold3 is distributed with licence conditions which restrict general purpose use. Boltz is a permissive, open-source re-implementation of AlphaFold3, but it is bottlenecked by increased VRAM requirements and requires high-end GPU hardware to model large molecular systems. Here we introduce Low Memory Inference for Boltz (LMI4Boltz) which reduces VRAM requirements using in-place updates, offloading tensors to host memory, careful management of functional scope and aggressive chunking of key operations. Using these strategies, LMI4Boltz increases the token size limit of Boltz-2 by 66.7% without sacrificing prediction accuracy. These optimisations improve the accessibility of Boltz using consumer grade hardware and unlock the ability to model large molecular systems. LMI4Boltz is available at https://github.com/tlitfin/lmi4boltz . Introduction The emergence of AlphaFold2 ( Jumper et al ., 2021 ) for protein structure prediction has initiated a paradigm shift in computational structural biology. High quality models have been generated for the entire UniProt database ( Varadi et al ., 2022 ) and dedicated efforts have focused on extending predictions to large scale homo- and hetero-oligomeric complexes ( Yu et al ., 2023 ). The combinatorial complexity of the potential interactome necessitates an efficient structure prediction model to enable comprehensive screening of candidate interactions. AlphaFold3 ( Abramson et al ., 2024 ) has substantially improved the efficiency of prediction workloads while also introducing the ability to model general molecular complexes. AlphaFold3 utilises a lightweight MSA processing module and a downstream diffusion process for generating atomic coordinates to reduce the computational requirements of structure prediction. AlphaFold3 is implemented in Jax and is ‘just in time’ compiled to ensure efficient hardware utilisation. However, the bespoke restrictions associated with the AlphaFold3 licence has motivated the development of open-source re-implementations with more permissive license conditions ( Wohlwend et al ., 2025 ; Liu et al ., 2024 ; Team et al ., 2025 ). Boltz is an AlphaFold3-style model for general purpose molecular structure prediction ( Wohlwend et al ., 2025 ; Passaro et al ., 2025 ). Compared with AphaFold3, Boltz reports competitive accuracy and computational efficiency. However, the Boltz implementation utilises the eager execution of PyTorch and does not benefit from integrated compiler optimisation of memory use. As a result, complex prediction size is limited to ∼1600 tokens using consumer grade hardware (24GB VRAM). To reconcile the memory disparity, we have introduced several optimisations to improve memory utilisation. For example, the pair representation is updated in-place to avoid temporary update tensors. In addition, several infrequently used tensors are offloaded to host memory to minimise baseline memory overhead. Functional scope is also carefully managed to avoid the temporary duplication of large tensors. Finally, after minimising memory utilisation, chunking is applied to additional model layers to mitigate newly emergent bottlenecks. Methods Inference settings The Boltz-2 baseline was run using the 744b4ae commit. max_msa_seqs was set to 4096 for consistency between Boltz-1 and Boltz-2 implementations. All modelling was conducted using expandable segments to allow flexible VRAM utilisation by the PyTorch caching allocator. VRAM requirement benchmark A human ubiquitin sequence (length: 76) was predicted by Boltz with the use_msa_server option. The alignment file was re-used to model an increasing number of ubiquitin subunits until each method registered an out-of-memory error. Experiments were run with a NVIDIA H200 GPU (141 GB VRAM) with per process memory fraction set to 0.17 to simulate 24GB VRAM capacity. Execution time was reported as the best of 3 replicates as recorded by the internal progress bar. Prediction accuracy benchmark Structures from the PDB test set were predicted by each of the proposed methods using the top-1 confidence output from 5 diffusion samples. The pre-generated Boltz-1 input files ( Wohlwend et al ., 2025 ) were used to maintain consistency with prior work. Prediction accuracy was evaluated by computing the lddt ( Mariani et al ., 2013 ) and TMscore (Zhang et al ., 2022) of model outputs against reference structures as implemented in OpenStructure v2.10 ( Biasini et al ., 2013 ). 8pe3 and 8t4r were excluded due to errors parsing the reference structures. In-place operations Updates to the pair representation are applied in-place to avoid creating a temporary update tensor. This strategy also maintains the output in a compact bfloat16 format. By contrast, Boltz-2 produces some intermediate tensors in full precision due to type promotion and auto-casting of layer norm outputs. Tensor offloading During inference, Boltz instantiates several objects with large memory footprints (e.g. the relative position encoding with size LxLx128, z_init with size LxLx128 and pdistogram with size LxLx64). These objects are used infrequently but persist in GPU VRAM once instantiated. Here we move these objects between host and GPU memory to minimise fixed memory overhead. Similarly, all layers with skip connections maintain parallel copies of the pair representation (LxLx128 size) while computing the update function. Here we push the MSA module skip connection to host memory while computing the required update to reduce peak GPU memory load. Function scope management In Boltz, the function scoping of the trifast implementation creates temporary duplicated copies of the large q, k, v tensors. In practice, this mitigates the theoretical savings of the trifast kernel by increasing VRAM requirements compared with naïve triangle attention. By managing function scope to maintain a single copy of the contiguous q, k, v tensors, the theoretical memory and throughput advantages can be realised. A similar strategy is also used to avoid duplication of the relative position encoding input tensor. Aggressive chunking We added command line parameters to allow interactive adjustment of chunk sizes to resolve bottlenecks as they emerged during benchmarking. In the extreme case, chunk_size_transition_z was set to 32 and chunk_size_tri_attn was set to 64. To minimize memory requirements, we extended the existing MSAFormer pair transition chunking strategy to the PairFormer and pair conditioning layers. Finally, we introduced new chunking functionality to the triangle multiplicative update (triangle_mult_gate_nchunks set to 4). Results Boltz-2 supports predictions with size up to ∼1600 tokens using a GPU with 24 GB VRAM ( Fig 1A, B ). By offloading key tensors to host memory and eliminating transient full precision operations in the model trunk (+memory), this capability can be increased by almost 50% ( Fig 1A, B ; >2356 tokens). Despite the overhead associated with moving data between devices, the execution time is also marginally improved when modelling 1596 tokens ( Fig 1C ). This is achieved by running additional operations in bfloat16 as well as reducing the burden on the caching allocator when approaching the device memory limit. Aggressively chunking operations at key memory bottlenecks (+chunk) further increases the length limit to >2660 tokens ( Fig 1A, B ) with an associated wall time cost of 8.5% for 1596 tokens (cf Boltz-2). It should also be noted that chunking is only required for large complexes and can be selectively disabled at smaller sizes (<2356 tokens) to maximise overall throughput. Download figure Open in new tab Figure 1. (A) Execution time required to predict the structure of an increasing number of ubiquitin subunits using each of the Boltz-2 implementations. (B) Maximum number of tokens able to be predicted before registering an out-of-memory error with 24GB VRAM. (C) Wall time required to predict the structure of 21 ubiquitin subunits (1596 tokens). (D) lddt of the PDB test set for LMI4Boltz compared with the original Boltz-2 implementation. LMI4Boltz is particularly efficient for specific sequence lengths which leads to peaks in observed execution times. For example, the time required to predict 2432 tokens is >6% less than the time for a smaller system with 2356 tokens. These optimum lengths coincide with precise multiples of 8 which maximises the efficiency of matrix multiplications on tensor core hardware. This observation highlights a potential strategy to improve throughput for large complexes by padding the length to a multiple of 8 during inference. However, realising this acceleration depends on the nature of the GPU hardware used for prediction. Despite several modifications, LMI4Boltz outputs closely match those generated by the canonical Boltz-2 implementation. In the PDB test set, the average lddt is 0.820 (cf. 0.820 for Boltz-2). Rare outlier cases occur when the confidence scores of output structures are similar enough that minor fluctuations change the top-1 ranked output ( Fig 1D ). Using TM-score as an assessment measure highlights outlier complexes which have different binding modes caused by changes in numerical precision (Fig S1A). However, the overall output quality remains un-affected since average TMscore is 0.841 (cf 0.840 for Boltz-2). Exact parity can be restored (Fig S1B) by minor modifications to Boltz-2 including skipping the trivial identity operation in the no dropout setting and casting the outputs of targeted operations to bfloat16 (otherwise promoted to fp32). In addition, since the Boltz-2 implementation of pair transition chunking is not isomorphic to the un-chunked version, the chunk_size_transition_z flag is disabled for LMI4Boltz to restore exact output parity. Discussion In addition to existing optimisations, LMI4Boltz can easily be extended to model larger molecular systems. For example, during triangle multiplicative updates, numerous copies of the pair representation are maintained in memory. These tensors could be juggled between GPU and host as required to relieve a memory bottleneck. However, frequent data movement within an internal loop can significantly increase the overall execution time. Here we have prioritised increasing the sequence length limit while maintaining throughput comparable to the original Boltz implementation. Alternative strategies such as the bespoke chunking strategy introduced in OpenFold ( Ahdritz et al ., 2024 ) may provide a path to relieve this bottleneck more efficiently. LMI4Boltz is primarily intended to support inference workloads and modifications such as in-place operations are not compatible with model training. However, the additional reduced precision operations can be directly adapted for training workloads. Offloaded tensors also provide a roadmap for rematerialisation to reduce memory requirements during training ( Chen et al ., 2016 ). In addition, the memory bottlenecks identified in this work could be optimised in future architectural updates. For example, the pairwise conditioner concatenates the relative position encoding with the trunk pair representation along the channel dimension which spikes memory use. This information can likely be combined more efficiently without impacting model accuracy. LMI4Boltz significantly increases the sequence length limit of Boltz by careful management of large intermediate tensors and chunking operations at key bottlenecks. While the optimum chunking strategy depends on the available hardware and the size of the prediction system, we demonstrate that robust low-memory inference can be achieved by aggressive chunking for only a small increase in computational cost. Despite minor variations in output compared with Boltz-2 associated with numerical precision, the differences do not negatively impact output quality. Overall, we expect LMI4Boltz to democratise access to the Boltz model by enabling execution using a wide range of consumer-grade hardware. Acknowledgements We gratefully acknowledge the support of the UNSW MWAC Structural Biology Facility and the UNSW Research Technology team for providing access to the computational infrastructure to support this research. Footnotes Corrected github link in abstract to https://github.com/tlitfin/lmi4boltz References 1. ↵ Abramson , J. et al. ( 2024 ) Accurate structure prediction of biomolecular interactions with AlphaFold 3 . Nature , 630 , 493 – 500 . OpenUrl CrossRef PubMed 2. ↵ Ahdritz , G. et al. ( 2024 ) OpenFold: retraining AlphaFold2 yields new insights into its learning mechanisms and capacity for generalization . Nat. Methods , 21 , 1514 – 1524 . OpenUrl CrossRef PubMed 3. ↵ Biasini , M. et al. ( 2013 ) OpenStructure: an integrated software framework for computational structural biology . Biol. Crystallogr ., 69 , 701 – 709 . OpenUrl CrossRef 4. ↵ Chen , T. et al. ( 2016 ) Training deep nets with sublinear memory cost . arXiv Prepr . arxiv: 1604.06174 . 5. ↵ Jumper , J. et al. ( 2021 ) Highly accurate protein structure prediction with AlphaFold . Nature , 596 , 583 – 589 . OpenUrl CrossRef PubMed 6. ↵ Liu , L. et al. ( 2024 ) Technical report of HelixFold3 for biomolecular structure prediction . arXiv Prepr . arxiv: 2408.16975 . 7. ↵ Mariani , V. et al. ( 2013 ) lDDT: a local superposition-free score for comparing protein structures and models using distance difference tests . Bioinformatics , 29 , 2722 – 2728 . OpenUrl CrossRef PubMed Web of Science 8. ↵ Passaro , S. et al. ( 2025 ) Boltz-2: Towards Accurate and Efficient Binding Affinity Prediction . bioRxiv , 2025.06.14.659707. 9. ↵ Team,B.A.M.L.A. et al . ( 2025 ) Protenix - Advancing Structure Prediction Through a Comprehensive AlphaFold3 Reproduction . bioRxiv , 2025.01.08.631967. 10. ↵ Varadi , M. et al. ( 2022 ) AlphaFold Protein Structure Database: massively expanding the structural coverage of protein-sequence space with high-accuracy models . Nucleic Acids Res ., 50 , D439 – D444 . OpenUrl CrossRef PubMed 11. ↵ Wohlwend , J. et al. ( 2025 ) Boltz-1: Democratizing Biomolecular Interaction Modeling . bioRxiv , 2024.11.19.624167. 12. ↵ Yu , D. et al. ( 2023 ) AlphaPulldown—a python package for protein–protein interaction screens using AlphaFold-Multimer . Bioinformatics , 39 , btac749 . OpenUrl CrossRef PubMed 13. Zhang , C. et al. ( 2022 ) US-align: universal structure alignments of proteins, nucleic acids, and macromolecular complexes . Nat. Methods , 19 , 1109 – 1115 . OpenUrl CrossRef PubMed View the discussion thread. Back to top Previous Next Posted November 01, 2025. Download PDF Supplementary Material Email Thank you for your interest in spreading the word about bioRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following LMI4Boltz: Optimising VRAM utilisation to predict large macromolecular complexes with consumer grade hardware Message Subject (Your Name) has forwarded a page to you from bioRxiv Message Body (Your Name) thought you would like to see this page from the bioRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share LMI4Boltz: Optimising VRAM utilisation to predict large macromolecular complexes with consumer grade hardware Thomas Litfin , Joshua Storm Caley , Katharine A. Michie bioRxiv 2025.10.29.684571; doi: https://doi.org/10.1101/2025.10.29.684571 Share This Article: Copy Citation Tools LMI4Boltz: Optimising VRAM utilisation to predict large macromolecular complexes with consumer grade hardware Thomas Litfin , Joshua Storm Caley , Katharine A. Michie bioRxiv 2025.10.29.684571; doi: https://doi.org/10.1101/2025.10.29.684571 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Bioinformatics Subject Areas All Articles Animal Behavior and Cognition (7629) Biochemistry (17660) Bioengineering (13881) Bioinformatics (41913) Biophysics (21436) Cancer Biology (18578) Cell Biology (25482) Clinical Trials (138) Developmental Biology (13372) Ecology (19890) Epidemiology (2067) Evolutionary Biology (24302) Genetics (15600) Genomics (22483) Immunology (17728) Microbiology (40365) Molecular Biology (17163) Neuroscience (88540) Paleontology (666) Pathology (2830) Pharmacology and Toxicology (4821) Physiology (7637) Plant Biology (15136) Scientific Communication and Education (2045) Synthetic Biology (4290) Systems Biology (9818) Zoology (2269)

Text is read by the "Ask this paper" AI Q&A widget below. Extraction quality varies by source — PMC NXML preserves structure cleanly, OA-HTML may include some navigation residue, and OA-PDF can have broken hyphenation. The publisher copy (via DOI) is the canonical version.

My notes (saved in your browser only)

Ask this paper AI returns verbatim quotes from the full text · source: preprint-html

Answers must be backed by verbatim quotes from this paper's full text. Hallucinated quotes are dropped automatically; if no verbatim passage answers the question, we say so. How this works

Citation neighborhood (no data yet)

We don't have any in-corpus citations linked to this paper yet. This is a recent paper (2025) — citers typically take a year or two to land, and the OpenAlex reference graph may still be filling in.

Source provenance

europepmc
last seen: 2026-05-20T01:45:00.602351+00:00
unpaywall
last seen: 2026-05-22T02:00:06.705733+00:00
License: CC-BY-4.0