Full text
25,504 characters
· extracted from
preprint-html
· click to expand
SimMS: A GPU-Accelerated Cosine Similarity implementation for Tandem Mass Spectrometry | bioRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-M677548'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search New Results SimMS: A GPU-Accelerated Cosine Similarity implementation for Tandem Mass Spectrometry Tornike Onoprishvili , Jui-Hung Yuan , Kamen Petrov , Vijay Ingalalli , Lila Khederlarian , Niklas Leuchtenmuller , Sona Chandra , Aurelien Duarte , Andreas Bender , View ORCID Profile Yoann Gloaguen doi: https://doi.org/10.1101/2024.07.24.605006 Tornike Onoprishvili 1 Independent consultant Find this author on Google Scholar Find this author on PubMed Search for this author on this site Jui-Hung Yuan 2 Pangea Botanica Germany GmbH , Hardenbergstrasse 32, 10623 Berlin, Germany Find this author on Google Scholar Find this author on PubMed Search for this author on this site Kamen Petrov 2 Pangea Botanica Germany GmbH , Hardenbergstrasse 32, 10623 Berlin, Germany Find this author on Google Scholar Find this author on PubMed Search for this author on this site Vijay Ingalalli 2 Pangea Botanica Germany GmbH , Hardenbergstrasse 32, 10623 Berlin, Germany Find this author on Google Scholar Find this author on PubMed Search for this author on this site Lila Khederlarian 3 Pangea Botanica Ltd , 15 Southampton Pl, London WC1A 2AJ, United Kingdom Find this author on Google Scholar Find this author on PubMed Search for this author on this site Niklas Leuchtenmuller 4 Wilde Ventures GmbH , In der Rehwiese 3, 40629 Düsseldorf, Germany Find this author on Google Scholar Find this author on PubMed Search for this author on this site Sona Chandra 2 Pangea Botanica Germany GmbH , Hardenbergstrasse 32, 10623 Berlin, Germany Find this author on Google Scholar Find this author on PubMed Search for this author on this site Aurelien Duarte 1 Independent consultant Find this author on Google Scholar Find this author on PubMed Search for this author on this site Andreas Bender 2 Pangea Botanica Germany GmbH , Hardenbergstrasse 32, 10623 Berlin, Germany Find this author on Google Scholar Find this author on PubMed Search for this author on this site Yoann Gloaguen 2 Pangea Botanica Germany GmbH , Hardenbergstrasse 32, 10623 Berlin, Germany Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Yoann Gloaguen For correspondence: yoann{at}pangeabio.com Abstract Full Text Info/History Metrics Supplementary material Data/Code Preview PDF Abstract Untargeted metabolomics involves a large-scale comparison of the fragmentation pattern of a mass spectrum against a database containing known spectra. Given the number of comparisons involved, this step can be time-consuming. In this work, we present a GPU-accelerated cosine similarity implementation for Tandem Mass Spectrometry (MS) with approximately 1000-fold speedup compared to the MatchMS reference at a rate of 0.005% incorrect matches and a rate of 0.002% incorrect scores. We describe the underlying reasons for these errors and provide means to avoid them. Introduction In the field of untargeted metabolomics, Tandem Mass Spectrometry (MS/MS) 1 is a well-established technique for identifying compounds within complex biological samples. The process works by comparing an unknown compound’s mass spectrum fragmentation pattern (“query”) against a database of known spectra (“reference”) in an effort to identify the unknown compound’s chemical composition 2 . Cosine similarity and its variants are popular methods facilitating MS/MS spectra comparison 3 , 4 . Cosine similarity works by calculating a cosine of the angle between vectors of fragmentation intensities (“peaks”). Since an exact match between two fragmentation spectra isn’t practically possible due to measurement errors, finding the cosine score involves solving an assignment problem between the two sets of spectral peaks, with the goal of finding a valid matching of peaks (within m/z tolerance) that maximize the value of the cosine. Calculating the exact cosine similarity (Hungarian cosine 5 ) is usually impractical for even moderate numbers of spectra. “Greedy” cosine similarity is an efficient approximation of the cosine similarity, which solves the assignment problem using a greedy heuristic 6 . “Modified” cosine similarity is an extension of greedy cosine that uses the precursor mass as an additional input and has been shown to outperform greedy cosine in some cases 3 . Huber et al . also introduced MatchMS 6 , an open-source MS/MS processing library in python. This library allows implementing easy-to-reproduce workflows to process raw mass spectral data into more useful forms (i.e. molecular networks). MatchMS conveniently implements all three types of cosine similarities. Unfortunately, while the MatchMS implementation is convenient, it is too slow for processing spectra on the scale of 10 10 pairwise comparisons or more. At such scales, which are routine in our untargeted metabolomics workflows, MatchMS requires tens of CPU-days, necessitating a search for more efficient and robust approaches 7 - 9 . To address this issue, Harwood et al . introduced BLINK 10 , which approximates the greedy cosine by first blurring the query spectral peaks, transforming it into a sparse matrix, and then performing sparse matrix multiplication between the query matrix and the reference matrix to directly compute the score matrix. This effectively side steps solving the peak assignment problem, allowing BLINK significant speed improvements over MatchMS. Unfortunately, BLINK rapidly loses accuracy when tolerance is larger than 10 −2 . In the current report, we speed up Cosine calculation from an engineering point of view. We leverage the fact that cosine calculation is readily parallelizable and rewrite both greedy and modified cosine algorithms using CUDA 11 into a single, highly optimized GPU kernel that, depending on underlying GPU hardware, can process spectra up to 1,000 times faster than MatchMS. We show that the approach exactly replicates MatchMS results and supports a much wider tolerance range without compromising accuracy. Further, we distribute the kernel into a user-friendly Python package that can act as a drop-in replacement for respective MatchMS cosine similarity classes. All code, results, and notebooks supporting this report are freely available under the MIT license at https://github.com/pangeAI/simms/ . Results We found that using the Cosine kernel implementation presented in this work with a single NVIDIA A100 GPU for calculations 12 is approximately 1,000x faster than using MatchMS for both greedy and modified cosine similarities, as shown in Figure 1 . Calculating greedy cosine similarity at the scale of 100,000 queries paired with 1.5 million reference spectra takes an estimated 13 weeks with MatchMS, in contrast to only 3 hours using the kernel. Download figure Open in new tab Figure 1: Cosine scoring runtimes for GNPS subsets with default parameters, with different GPUs and methods. For example, the red line shows the SimMS performance on A100 for modified cosine scoring, and the brown line is the same calculation with MatchMS. Marker dots (A) represent real measurements, and dashed lines (B) are their linear interpolations. (C) is a marker of 1.5·10 11 comparisons, and represents our target processing goals. Notably, all approaches seem to have a similar asymptotic computational complexity. Modified cosine similarity is slower than greedy cosine (also shown in Figure 1 ), independent of hardware used, since the steps required to calculate the modified cosine score includes all steps required for calculating the greedy cosine score 3 . We performed a direct comparison of results for predicted scores and the number of matches, as shown in Figure 2 . We found that for default parameters, the kernel and MatchMS results are within ±0.001 of each other 99.99% of the time. Since the kernel is algorithmically equivalent to its MatchMS counterpart, the few errors stem from peaks that are almost exactly tolerance distance apart. In other words, when paired with MatchMS, these peaks appear to be inside tolerance distance, but processing them with a GPU changes their binary floating point representation, which is enough to make them appear outside of the tolerance distance. When a spectrum has a single very intense peak, such binary representation changes can result in large score errors. The score comparison plots in panels Figure 2A and Figure 2C show these kinds of errors in the bottom right. Download figure Open in new tab Figure 2: Direct comparison of scores of 4.1M random GNPS spectra pairs, illustrating SimMS error patterns. Red dots denote incorrect (absolute error > 0.001) SimMS outputs. (A) and (B) show greedy cosine results between SimMS and MatchMS, performed at float32 precision. In (C) and (D) we compare modified cosine results at float64 precision. Occasional large errors in (A) and (C) are caused by spectra with very few (one or two) large peaks. The rate of incorrect scores is usually lower than the rate of incorrect number of matches. This asymmetry originates from the filtering step. Filtering usually removes most of the low-intensity matches from the score calculation, thus making the missing or extra match unlikely to affect the score value. In Figure 3 we can see how the changing tolerance and match limit influence SimMS performance. In Figure 3A and 3D we see that lowering these values can significantly speed up spectra processing time, but this comes at the price of increasing the overflow rate, as seen in Figure 3B and 3E . The accuracy rate in Figure 3C and 3F is approximately equal to the 1 - overflow rate. Download figure Open in new tab Figure 3: Key metrics as a function of changing tolerance (A) - (C) and match limit (D) - (F). In a and d speed means the number of comparisons per second. “Overflow” is the proportion of scores that are returned with the overflow flag. Accuracy is the proportion of scores that are within ±0.001 of MatchMS. Methods Implementation We used as input a list of references and queries, denoting their respective lengths as R and Q . For both lists, consecutive spectra are grouped into batches of size B , the last batch contains leftover spectra and padding, as needed. Inside each batch, all spectra are concatenated into a single ℝ 2, B, M tensor, where M is the number of peaks in the longest spectrum in that batch, B is the batch size ( Figure 4a ). A batch contains stacked peak m/z and intensity values in the first dimension. Spectra that are shorter than M are padded with zeros. Spectra that are larger than M are truncated with an argument N max peaks . Download figure Open in new tab Figure 4: Overview of processing on GPU. In a first step (A), we pack all spectra and metadata in 3D and 2D tensors, respectively. In a second step (B), we align the spectra with the GPU grid. Steps (C) and (D) happen within a single GPU thread. In (C), we accumulate potential matching peaks up to the given match limit. In the last step (D), we sort, deduplicate and reduce matched peaks, returning three values. The “metadata” tensor is created alongside the spectra batch. The metadata tensor contains the length, cosine norm, and, in case of modified cosine, the precursor m/z values. Metadata tensor is a ℝ K,B tensor, where K is either 4 or 6, and B is the batch size. For greedy cosine, dimension K is 4 and consists of lengths of reference and query spectra and norms of reference and query spectra. In case of modified cosine metadata additionally contains precursor m/z values for reference and query spectra. For batches corresponding to leftover spectra, we pad the resulting empty space with zeros. The full similarity matrix of size R × Q is infeasible to store in GPU memory. Therefore processing is done block-by-block. The full R × Q grid is split into B × B sized, non-overlapping blocks. In total this results in number of blocks. For each block an output ℝ 3, B,B tensor is allocated on the GPU. A pair of batches of spectra, and their respective metadata tensors are transferred to the GPU. The tolerance, m/z power, intensity power, N match limit and N max peaks are supplied to the kernel as compilation-time constants. At this point the kernel is launched. The kernel is written using Numba 13 . Inside the kernel, a single CUDA thread is assigned to calculate the cosine score between one reference spectrum and one query spectrum from the supplied batches. The computing power of SimMS stems from the fact that modern GPUs can process tens of thousands of threads simultaneously, allowing us to process each block in a fraction of a second. The kernel itself consists of three main stages ( Figure 4C and 4D ). First, pairs of peaks within tolerance are collected. A maximum of N match limit pairs are collected. If the number of pairs exceeds this limit, an overflow flag is raised and the collection stops early. Next, the pairs are sorted by the value of the product of the intensities of paired peaks. Finally, the ordered pairs are deduplicated and accumulated into an unnormalized cosine score. An auxiliary boolean array is used to mark each accumulated peak, to avoid duplicate contributions to the score. As a final step, the two norms from metadata are multiplied to obtain the normalizing constant, which divides the unnormalized score to produce the final cosine score ( Figure 4D ). After the execution, the block output tensor contains 3 results for every reference/query pair in the batch. These are: score, number of matches, and overflow (binary flag). Each block is then concatenated together in order to form the full R × Q similarity matrix. In case of processing a very large set of references and queries (larger than 100,000), the required memory to store the full similarity matrix as a dense array is impractically large. Additionally, we find that most of the scores are lower than 0.5. For such cases we use “sparse” implementation, where the similarity matrix is filtered to discard all results with score below a user-defined “sparse threshold” and then store the remaining entries as a sparse matrix in DOK format. Finally, we concatenate all the results and return them, either as a dense array or as a sparse DOK matrix. Hardware All of our kernel and MatchMS experiments were performed on rented Vastai (Vast.ai) instances. Preferably, we rented instances with at least 16GB RAM, 8 or more CPUs, and at least a single RTX4090 GPU. Performance of the original MatchMS algorithm is independent from GPU, while our kernel performance heavily depends on it - A100 GPU usually outperforms RTX4090, which in turn outperforms older GPUs. Interoperability We have taken care to make the SimMS package fully compatible with MatchMS. During the implementation of the kernel, we also discovered that Cosine Greedy scores from MatchMS were semi-randomly fluctuating when run on different CPU hardware, causing the exact MatchMS results to be unrepeatable. We later patched this bug, and from version 0.24.0 onwards, the patch has been merged into the core MatchMS package. Correction details are available at https://github.com/matchms/matchms/pull/596 . The full code is available on GitHub at https://github.com/PangeAI/simms under an MIT license. Footnotes https://github.com/PangeAI/SimMS References 1. ↵ Dettmer Katja , Aronov Pavel A. , and Hammock Bruce D. 2007 . “ Mass Spectrometry-based Metabolomics .” Mass Spectrometry Reviews 26 ( 1 ): 51 – 78 . doi: 10.1002/mas.20108 . OpenUrl CrossRef PubMed Web of Science 2. ↵ Wout Bittremieux , Wang Mingxun , and Dorrestein Pieter C. 2022 . “ The Critical Role That Spectral Libraries Play in Capturing the Metabolomics Community Knowledge .” Metabolomics 18 ( 12 ): 94 . doi: 10.1007/s11306-022-01947-y . OpenUrl CrossRef 3. ↵ Watrous , J. et al. Mass spectral molecular networking of living microbial colonies . Proc. Natl. Acad. Sci . 109 , E1743 – E1752 ( 2012 ). doi: 10.1073/pnas.1203689109 OpenUrl Abstract / FREE Full Text 4. ↵ Schollée Jennifer E. , Schymanski Emma L. , Stravs Michael A. , Gulde Rebekka , Thomaidis Nikolaos S. , and Hollender Juliane . 2017 . “ Similarity of High-Resolution Tandem Mass Spectrometry Spectra of Structurally Related Micropollutants and Transformation Products .” Journal of the American Society for Mass Spectrometry 28 ( 12 ): 2692 – 2704 . doi: 10.1007/s13361-017-1797-6 . OpenUrl CrossRef 5. ↵ Munkres , James . 1957 . “ Algorithms for the Assignment and Transportation Problems .” Journal of the Society for Industrial and Applied Mathematics 5 ( 1 ): 32 – 38 . doi: 10.1137/0105003 . OpenUrl CrossRef 6. ↵ Huber F , Verhoeven S , Meijer C , Spreeuw H , Castilla EMV , Geng C , et al. 2020 . Matchms -processing and similarity evaluation of mass spectrometry data . Journal of Open Source Software , 5 ( 52 ), 2411 , doi: 10.21105/joss.02411 OpenUrl CrossRef 7. ↵ Horai , H. , Arita , M. , Kanaya , S. , Nihei , Y. , Ikeda , T. , Suwa , K. , Ojima , Y. , Tanaka , K. , Tanaka , S. , Aoshima , K. , Oda , Y. , Kakazu , Y. , Kusano , M. , Tohge , T. , Matsuda , F. , Sawada , Y. , Hirai , M.Y. , Nakanishi , H. , Ikeda , K. , Akimoto , N. , Maoka , T. , Takahashi , H. , Ara , T. , Sakurai , N. , Suzuki , H. , Shibata , D. , Neumann , S. , Iida , T. , Tanaka , K. , Funatsu , K. , Matsuura , F. , Soga , T. , Taguchi , R. , Saito , K. and Nishioka , T. 2010 . MassBank: a public repository for sharing mass spectral data for life sciences . J. Mass Spectrom ., 45 : 703 – 714 . doi: 10.1002/jms.1777 OpenUrl CrossRef PubMed Web of Science 8. Guijas C , Montenegro-Burke JR , Domingo-Almenara X , Palermo A , Warth B , Hermann G , et al. 2018 . METLIN: A Technology Platform for Identifying Knowns and Unknowns . Anal Chem . 90 : 3156 – 3164 . doi: 10.1021/acs.analchem.7b04424 OpenUrl CrossRef PubMed 9. ↵ Wang , M. , Carver , J. , Phelan , V. et al. 2016 . Sharing and community curation of mass spectrometry data with Global Natural Products Social Molecular Networking . Nat Biotechnol 34 , 828 – 837 . doi: 10.1038/nbt.3597 OpenUrl CrossRef PubMed 10. ↵ Harwood T.V. , Treen D.G.C. , Wang M. et al. 2023 . BLINK enables ultrafast tandem mass spectrometry cosine similarity scoring . Sci Rep 13 , 13462 . doi: 10.1038/s41598-023-40496-9 OpenUrl CrossRef 11. ↵ Nvidia , Vingelmann P. , & Fitzek F. H. P. 2020 . CUDA, release: 10.2.89 . Retrieved from https://developer.nvidia.com/cuda-toolkit 12. ↵ Choquette J. , Gandhi W. , Giroux O. , Stam N. , and Krashinsky R. 2021 . NVIDIA A100 Tensor Core GPU: Performance and Innovation , in IEEE Micro , vol. 41 , no. 2 , pp. 29 – 35 . , doi: 10.1109/MM.2021.3061394 . OpenUrl CrossRef 13. ↵ Lam SK , Pitrou A , Seibert S. Numba: a LLVM-based Python JIT compiler . Proceedings of the Second Workshop on the LLVM Compiler Infrastructure in HPC . Austin, Texas : Association for Computing Machinery ; 2015 . pp. 1 – 6 . doi: 10.1145/2833157.2833162 OpenUrl CrossRef Back to top Previous Next Posted July 25, 2024. Download PDF Supplementary Material Data/Code Email Thank you for your interest in spreading the word about bioRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following SimMS: A GPU-Accelerated Cosine Similarity implementation for Tandem Mass Spectrometry Message Subject (Your Name) has forwarded a page to you from bioRxiv Message Body (Your Name) thought you would like to see this page from the bioRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share SimMS: A GPU-Accelerated Cosine Similarity implementation for Tandem Mass Spectrometry Tornike Onoprishvili , Jui-Hung Yuan , Kamen Petrov , Vijay Ingalalli , Lila Khederlarian , Niklas Leuchtenmuller , Sona Chandra , Aurelien Duarte , Andreas Bender , Yoann Gloaguen bioRxiv 2024.07.24.605006; doi: https://doi.org/10.1101/2024.07.24.605006 Share This Article: Copy Citation Tools SimMS: A GPU-Accelerated Cosine Similarity implementation for Tandem Mass Spectrometry Tornike Onoprishvili , Jui-Hung Yuan , Kamen Petrov , Vijay Ingalalli , Lila Khederlarian , Niklas Leuchtenmuller , Sona Chandra , Aurelien Duarte , Andreas Bender , Yoann Gloaguen bioRxiv 2024.07.24.605006; doi: https://doi.org/10.1101/2024.07.24.605006 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Bioinformatics Subject Areas All Articles Animal Behavior and Cognition (7799) Biochemistry (18204) Bioengineering (14382) Bioinformatics (43096) Biophysics (21970) Cancer Biology (19100) Cell Biology (26174) Clinical Trials (138) Developmental Biology (13646) Ecology (20385) Epidemiology (2067) Evolutionary Biology (24844) Genetics (15843) Genomics (22976) Immunology (18198) Microbiology (41340) Molecular Biology (17538) Neuroscience (90815) Paleontology (679) Pathology (2904) Pharmacology and Toxicology (4948) Physiology (7869) Plant Biology (15504) Scientific Communication and Education (2066) Synthetic Biology (4431) Systems Biology (10008) Zoology (2314) (function(){function c(){var b=a.contentDocument||a.contentWindow.document;if(b){var d=b.createElement('script');d.innerHTML="window.__CF$cv$params={r:'a1c5acf36b3a599e',t:'MTc4NDI1Mzg2Mg=='};var a=document.createElement('script');a.src='/cdn-cgi/challenge-platform/scripts/jsd/main.js';document.getElementsByTagName('head')[0].appendChild(a);";b.getElementsByTagName('head')[0].appendChild(d)}}if(document.body){var a=document.createElement('iframe');a.height=1;a.width=1;a.style.position='absolute';a.style.top=0;a.style.left=0;a.style.border='none';a.style.visibility='hidden';document.body.appendChild(a);if('loading'!==document.readyState)c();else if(window.addEventListener)document.addEventListener('DOMContentLoaded',c);else{var e=document.onreadystatechange||function(){};document.onreadystatechange=function(b){e(b);'loading'!==document.readyState&&(document.onreadystatechange=e,c())}}}})();
Text is read by the "Ask this paper" AI Q&A widget below.
Extraction quality varies by source — PMC NXML preserves structure
cleanly, OA-HTML may include some navigation residue, and OA-PDF can
have broken hyphenation. The publisher copy
(via DOI)
is the canonical version.