Comparing the performance of functional versus taxonomic metagenomics for detecting ammonia disturbances in the biogas system

preprint OA: closed CC-BY-4.0
📄 Open PDF Full text JSON View at publisher

Abstract

ABSTRACT Biogas is a renewable energy source with great potential, but its production is frequently hindered by process disturbances, of which a high ammonia concentration is one common cause. It is desirable that such disturbances are found as early as possible, and metagenomics data has the potential to improve this detection. This study compares functional and taxonomic aspects of metagenomics data, hypothesizing that functional data will perform better for detecting ammonia disturbances. The hypothesis was tested by metagenomic sequencing of samples from three independent studies, which followed lab-scale reactors during ammonia disturbances. The resulting sequences were used to predict protein-coding genes, which were functionally and taxonomically annotated. The read counts of these features were fitted to disturbance states and ammonia concentrations of reactor samples using regularized regression, which allowed filtering out irrelevant features even when the number of features was much larger than the number of samples. Taxonomic data had similar or better performance in detecting ammonia disturbances and in fitting ammonia concentrations, both when analyzing separate studies as well as when analyzing the combined data of the studies. Our hypothesis that functional metagenomics would outperform taxonomic metagenomics was therefore not supported. Visual abstract
Full text 54,275 characters · extracted from preprint-html · click to expand
Comparing the performance of functional versus taxonomic metagenomics for detecting ammonia disturbances in the biogas system | bioRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-M677548'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search New Results Comparing the performance of functional versus taxonomic metagenomics for detecting ammonia disturbances in the biogas system View ORCID Profile Dries Boers , View ORCID Profile Olivier Chapleur , View ORCID Profile Anders F Andersson , View ORCID Profile Anna Schnürer doi: https://doi.org/10.1101/2025.06.10.658511 Dries Boers 1 Dept. of Molecular Sciences, Swedish University of Agriculture (SLU) , Uppsala, Sweden Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Dries Boers For correspondence: dries.boers{at}slu.se anna.schnurer{at}slu.se Olivier Chapleur 2 PROSE, Paris-Saclay University, National Research Institute for Agriculture, Food and Environment (INRAE) , Antony, France Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Olivier Chapleur Anders F Andersson 3 Department of Gene Technology, Science for Life Laboratory, KTH Royal Institute of Technology , Stockholm, Sweden Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Anders F Andersson Anna Schnürer 1 Dept. of Molecular Sciences, Swedish University of Agriculture (SLU) , Uppsala, Sweden Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Anna Schnürer For correspondence: dries.boers{at}slu.se anna.schnurer{at}slu.se Abstract Full Text Info/History Metrics Supplementary material Preview PDF ABSTRACT Biogas is a renewable energy source with great potential, but its production is frequently hindered by process disturbances, of which a high ammonia concentration is one common cause. It is desirable that such disturbances are found as early as possible, and metagenomics data has the potential to improve this detection. This study compares functional and taxonomic aspects of metagenomics data, hypothesizing that functional data will perform better for detecting ammonia disturbances. The hypothesis was tested by metagenomic sequencing of samples from three independent studies, which followed lab-scale reactors during ammonia disturbances. The resulting sequences were used to predict protein-coding genes, which were functionally and taxonomically annotated. The read counts of these features were fitted to disturbance states and ammonia concentrations of reactor samples using regularized regression, which allowed filtering out irrelevant features even when the number of features was much larger than the number of samples. Taxonomic data had similar or better performance in detecting ammonia disturbances and in fitting ammonia concentrations, both when analyzing separate studies as well as when analyzing the combined data of the studies. Our hypothesis that functional metagenomics would outperform taxonomic metagenomics was therefore not supported. Download figure Open in new tab Introduction The biogas system and ammonia disturbances Biogas is an energy source which can be produced carbon-neutrally while handling society’s biodegradable waste streams. Its purified form, biomethane, can replace fossil methane gas, and anaerobic digestion, the process in which biogas is produced, allows for the recycling of the nutrients (nitrogen, phosphorous) in the input waste streams[ 1 ]. These properties make biogas an attractive energy source, and the European Union has dedicated itself to upscaling its production of biogas in the REPowerEU plan[ 2 ]. A challenge in biogas production is that the process is sensitive to process disturbances, which lead to reduced biogas production, and in severe cases even process failure[ 3 ]. Such disturbances have been shown to occur frequently, lasting from weeks up to months, with methane output decreasing by 30%[ 4 ]. As a preventive measure, most biogas plants are being operated at lower organic loading rate than needed for reaching optimal biogas production[ 3 ]. A common cause for disturbances of the anaerobic digestion process is ammonia. Anaerobic digestion consists of interlocking chains of microbial metabolism. The final chain in this network are the methanogens, methane-producing archaea, and these are particularly sensitive to ammonia poisoning[ 5 ]. The cellular mechanism of this inhibition is not clear, but the ability of ‘free’ ammonia (NH 3 ) to permeate across membranes seems to be involved, because toxicity is lower at lower pH, at which more ammonia is converted to ammonium ions (NH 4 + ), which are not freely-diffusing due to their charge[ 6 ]. When the methanogens are inhibited, this leads to the accumulation of upstream metabolites, such as hydrogen and carbon dioxide, and also acetate and other volatile fatty acids (VFAs)[ 7 ]. Such build-ups are therefore clear signals of a disturbance. Towards microbial community monitoring Because of the cost of disturbances in biogas plants, the process parameters of anaerobic digestion are often intensively monitored. In a review of such monitoring, it was shown that volatile fatty acids (VFAs) and biogas composition are indicators of disturbances at an early stage[ 3 ]. However, it has been stated that including data on the microbial community could improve monitoring effectiveness[ 8 , 9 ], as such data might reveal disturbances at an earlier stage than only chemical parameters[ 9 ], and because models based upon chemical parameters may stop working after large microbiological shifts[ 10 ]. The state of a microbial community can be defined by analyzing its genetic sequences. Two widely-used techniques are available for this: amplicon sequencing, which targets specific taxonomic marker sequences, and whole-genome ‘shotgun’ sequencing, which randomly samples sequences from all genomes in a sample. This study mainly concerns itself with data generated by the latter technique, and refers to it as metagenomic data. After sequence processing, such data can be analysed from two perspectives. First, the taxonomic perspective views ‘who’ is present in a sample, that is, which taxa are present in an environment. Second, the functional perspective regards which genetic functions are present and which metabolic pathways are encoded. The two alternative perspectives raise the question which of them reflects the state of a microbial system best. Comparing the taxonomic and functional perspectives Intuitively, the functional perspective looks more promising than the taxonomic perspective, as functions could build more fine-grained models of the biogas system. This intuition is further supported upon the concept of functional redundancy, in which taxa can be replaced by other taxa with identical functions without any effect on the system as a whole[ 11 ]. Furthermore, microbial traits can vary greatly between closely-related organisms, even within species[ 12 ]. A phenomenon that also supports the functional perspective is that traits can be transmitted to distant taxa in horizontal gene-transfer, although it has been proposed that such transfer mostly concerns simple traits such as antibiotic resistance, while complex traits such as methanogenesis are strongly coupled to taxonomy[ 13 ]. Findings of whether the functional or taxonomic perspective is superior have varied. In the general microbiological discussion, Xu et al .[ 14 ] showed that in the Human Microbiome Project, the difference in classification accuracy using taxonomic 16S data and metagenomic functional annotation was not significant. However, in a landmark paper of the microbial biogeography field, Louca et al .[ 15 ] showed that in the invariable micro-environments of rain-forest beaker plants, functional profiles predicted from 16S data were more constant than that 16S itself, which led them to suggest that functional data in their field should be “the baseline […], particularly when the ultimate focus is on ecosystem functioning”. Again in a medical context, Casimiro-Soriguer et al . [ 16 ] compared the use of functional and taxonomic profiles to predict colorectal cancer based upon fecal metagenomes from seven independent studies, finding that taxonomic data generally performed best. In summary, whether functional or taxonomic features give better performance is still “subject to debate”[ 17 ]. Even though many studies have investigated the microbial composition and activity in biogas systems, the functional perspective has rarely been compared to the taxonomic perspective. For example, Campanaro et al .[ 18 ] looked at taxonomy and function of the assembled metagenomes of different full-scale biogas plants, but focused on correlation of taxonomy to process parameters. In another study, Fischer et al .[ 19 ] followed inoculates in batch experiments, and in half of these ammonia was increased. However, while 16S profiles and transcriptomics were both analyzed, their information content was not compared. A study by Lin et al .[ 20 ] may be the most informative for the question in mind; they followed nine reactors over time to study the predictability of response to changes in feeding. They found that this change was both reflected in an altered taxonomy composition and in (functional) metatranscriptomic data. However, they did not compare which of these perspectives followed the change best. Aim and hypothesis In this study, we aim to assess whether functional metagenomic data is more accurate than taxonomic data in representing the state of microbiological systems, specifically in the case of biogas reactors which are disturbed due to changing ammonia concentrations. The goal of this study is to contribute to the function versus taxonomy debate described above, both in general and in the specific context of the biogas system. Our hypothesis is that functional analysis will perform better than taxonomic analysis. To emphasize the confirmatory approach of this study, the hypothesis has been ‘preregistered’[ 21 ] in a document which was digitally signed before starting with data analysis (Supplementary Document 1).The study was further expanded to include a taxonomic perspective based upon 16S rRNA amplicon-based data. Amplicon data is widely used, and is generally less expensive to generate than shotgun sequencing data, so including a comparison of the two types is of general interest. By testing this hypothesis within the biogas system, we intend to explore the possibility for improving its monitoring, both by reassessing the performance of the taxonomy perspective, which has been used in biogas research more frequently, and by including the functional perspective, which is less explored. Methods Sample selection and classification Three independent studies were identified which all followed ammonia-induced disturbances over time in multiple lab-scale, mesophilic (37 °C) semi-continuous stirred-tank reactors. These studies are referred to as the Lemaigre study[ 10 ], the Cardona study[ 22 ] and the Ahrens study[ 23 ], and their characteristics are summarized in Table 1 . View this table: View inline View popup Download powerpoint Table 1. Description of included studies. For each of these studies, reactor performance parameters over time were analyzed; these included feeding parameters, gas production, free ammonia nitrogen (FAN) concentrations, VFA concentrations and 16S profiles. Based upon this analysis, a selection for metagenomics sequencing was made of available samples. The selected samples were also classified into ‘disturbed’ and ‘undisturbed’ classes based upon the same parameters; especially their VFA concentrations. DNA was extracted from the selected samples using protocols that differed between the studies. For the Lemaigre study, new extractions were made using the AllPrep PowerViral DNA/RNA Kit[ 24 ]. For the Cardona study, DNA was used that had been extracted for their original study using the PowerSoil DNA Isolation Kit[ 25 ] and which had been stored at -80 °C. For the Ahrens study, extractions were made using the FastDNA Spin Kit for Soil[ 26 ]. The resulting DNA extracts were sent to Eurofins Genomics for sequencing (Illumina Novaseq, paired end, >10 M reads per sample, 150 bp per read). Sequence processing The resulting sequences were processed using an adapted version of the Snakemake metagenomics workflow nbis-meta[ 27 ]. The workflow started with standard read processing; that is, trimming using Trimmomatic[ 28 ], quality-checking using FastQC[ 29 ] and MultiQC[ 30 ], assembly into contigs (each sample individually) using Megahit[ 31 ], read alignment to their respective assembly using Bowtie 2[ 32 ] and duplicate removal using SAMtools[ 33 ]. The further processing of contigs is described in greater detail, as their functional and taxonomic annotation is of specific interest to this publication. Protein-coding genes were predicted within contigs by Prodigal[ 34 ]. For functional annotation, eggNOG-mapper 2[ 35 ] mapped the amino acid sequences of the predicted genes to the eggNOG database version 5[ 36 ], using DIAMOND[ 37 ] for alignment, and then used this alignment to assign eggNOG orthologous groups (eggNOG_OGs). eggNOG was used because it is the largest database for orthologs; it is automatically inferred based upon evolutionary patterns. Its size decreases the risk of missing functions of understudied micro-organisms, which have been suggested to be abundant in the biogas system[ 38 ]. Besides the assigned eggNOG_OG, KEGG orthology data[ 39 ] included from the eggNOG database was used for additional functional annotation levels. For taxonomic annotation, predicted protein-coding gene amino acid sequences were aligned by CAT[ 40 ], also using DIAMOND, against the NCBI non-redundant protein database[ 41 ], after which the last common ancestor of the matches with a low E-value and large sequence similarity was determined based upon the NCBI taxonomy database[ 41 ]. It should be noted that the original implementation of CAT used a majority vote per contig, which is not included in this analysis; therefore, the current procedure is referred to as ‘gene-level annotation of taxonomy’ (GAT). Finally, functional and taxonomic features were quantified by counting the reads mapping to every protein-coding gene predicted by Prodigal and summing these per feature. To include amplicon sequencing data in the comparison, 16S rRNA sequencing counts were included from the original studies. These sequencing counts were based upon different sequencing protocols and also on different methods for inferring sequence clusters (using amplicon sequence variants or operational taxonomic units) and different taxonomy databases. Importantly, GAT and 16S data types contain a hierarchical component, that is, organisms which differ at species rank may be the same when viewed at domain rank. A similar hierarchy holds for KEGG, because in that database, counts can be analyzed per ortholog (the lowest level), per module, or pathway level (the highest level). Statistical analysis Statistical analysis of the functional and taxonomic feature counts was performed in R[ 42 ] using packages from the tidyverse[ 43 ]. See Fig. 2 for a visual summary of the analysis workflow. First, the read counts for all features were transformed per sample, using the centered log ratio from package vegan[ 44 ], to compensate for a compositionality bias[ 45 ]. The counts were then centered per feature, but they were not scaled, as larger differences in a feature should give it larger weight in subsequent analyses. Regularized regression was then applied to the transformed features count data. Regularization makes large and complex models smaller and simpler by removing features, which is necessary when the number of features (far) outnumbers the number of samples in a dataset. The package glmnet[ 46 ] was used, and elastic net models were constructed using mixing parameter α = ½. The elastic net uses the regularization parameter ‘λ’ to apply lasso-like penalties based upon the number of variables, resulting in the removal of features that affect model performance least, and uses the same λ to apply ridge-like penalties based upon the sum of squared coefficients. The combination of the two types of penalties circumvents the problem of solution instability due to multicollinearity, which is incurred when only applying lasso penalization[ 47 ]. Two distinct types of regularized regression were applied. In logistic regularized regression, feature data was fit to the disturbed versus undisturbed classification. In linear regularized regression, feature data was fit to FAN concentrations, which are the primary cause of the disturbance of these reactors. Both of these data types had added value, because while the classification dataset can integrate multiple data types in a simple, explainable way, so-called separation can make the performance of models with more features than samples meaningless[ 48 ]. Linear regression is not hindered by the same phenomenon. As negative measures of performance (loss functions) of the regularized models, the defaults for logistic and linear regression were used: deviance and mean squared error (MSE), respectively. Mean loss was computed for every value of λ, using leave-one-out cross-validation, and these means were plotted against the number of features of these models. These numbers of features are inversely related to λ, because as models become more regularized, their number of features decreases. Models with minimal mean loss and, secondarily, a minimal number of included features were selected from the fitted models. In linear regression, such minimal mean loss models could be selected directly, while in logistic regression, loss tends to decrease as models grow larger, and therefore the smallest models with near-minimal loss were selected manually instead. In data types with multiple hierarchical levels, these minimal loss models were used to determine and select the best-performing subtype. Superkingdom level for GAT data and domain and kingdom level for 16S data were not selected however, because their small number of features would likely lead to overfitting in regularized regression. The minimal mean loss models of the different data types, functional and taxonomic, were then compared to one another. The above steps were first applied to the separate study datasets. Consecutively, a combined dataset was built from the feature counts of the three studies and this was submitted to the same regularized regression procedure as above. Finally, to get more insight in the regularized models, the features of the minimal loss models were extracted. Results Sample selection and classification Reactor parameters were collected from three independent studies, in which biogas reactors encountering ammonia disturbances were followed over time. Three of these parameters are presented in Fig. 3 , and the feeding parameters in Suppl. Table 1. In all three studies, concentrations of FAN changed over time in all reactors (except for the control reactor of the Lemaigre study). For the Lemaigre and Cardona studies, the changes were induced as part of their study design, while for the Ahrens study, this occurred due to high protein content in the substrate. Coincidentally with the changes in the FAN concentrations, a disturbance occurred, that is, the production of methane decreased, and VFA concentrations increased. The disturbances were followed by recovery events with methane production returning to initial values and VFA concentrations decreasing. The magnitude of VFA concentration peaks varied between the studies, but VFA accumulation was in all cases coupled to a decreased methane production, illustrating the imbalance in the microbial community metabolism. The duration of the disturbance, as predicted based on the VFA levels, varied between the studies. Expressed in hydraulic retention time (HRT), which quantifies how long a volume remains in the reactor, the Ahrens study showed the fastest recovery time (approx. 50 days, 1 HRT), while the Lemaigre and Cardona studies took longer to recover (both at least 100 days, which respectively corresponded to 2 and 4 HRTs). In the Ahrens study, the relative levels of change of the reactors’ FAN levels were also much smaller than those of the other studies, and their methane production never fully halted. Time points for the Lemaigre and Cardona studies consisted of time points at the ends of phases of their feeding regimes as described in their publications, which generally matched VFA profiles. For the Cardona study, day 73 and 147 were also added, because in a principal component analysis based upon 16S data ( Figure 4 of that study[ 22 ]), these samples fell outside the three sample clusters to which (end-of-feeding-phase time points) day 70, 98, 127 and 175 belong. For the Ahrens study, time points for metagenomic sequencing were selected based upon VFA concentration profiles. The classification of all time points as disturbed or undisturbed (as indicated in Fig. 3 ) was primarily based upon VFA concentrations for all studies. Day 73 of the Cardona study and day 43 of the Ahrens study were left unclassified however, and left out of further analysis, as their status was considered unclear. The same was the case for day 239 of the Lemaigre study and 141 of the Ahrens study, because they were taken after disturbance, which neither corresponded to the undisturbed state, nor to the disturbed state. Supervised analysis: regularized regression The metagenomic sequencing data was processed and annotated, resulting in functional and taxonomic count data, which was combined with taxonomic 16S amplicon count data. The different annotation variants of count data were used to build models; logistic regression models were used to predict samples’ disturbance status; linear regression models were used to estimate FAN concentrations, which was considered to be the principal cause of the disturbance. For both types of regression analysis, each dataset was analyzed at multiple, hierarchical levels. For example, metagenomes could be annotated at different taxonomic ranks when using the GAT tool; the same was true for 16S sequences. When annotating a metagenome functionally using the KEGG database, this could be done at the lowest, single ortholog level, but also on a higher, module level. For each count type, per study, the performances in regularized logistic regression and linear regression per hierarchy level (Suppl. Fig. 1-3) were compared to determine an optimal level. The number of variants for each count type, per study (Suppl. Table 2-4) were also assessed, to confirm correct data labeling. Download figure Open in new tab Figure 1. Workflow diagram of the (shotgun) metagenomics data processing to functional counts, in blue, and taxonomic counts, in light-red. GAT: gene-level annotation of taxonomy. Download figure Open in new tab Figure 2. Workflow diagram of the statistical analysis of feature counts. CLR: centered log-ratio transformation. FAN: free ammonia nitrogen. Download figure Open in new tab Figure 3. Reactor parameters for all three studies over time. In Lemaigre panels, dashed line represents control reactor without addition of extra ammonium. Breaks on time-axis indicate selected sampled time points. Blue labels indicate time points classified as disturbed, black labels with large digits indicate undisturbed time points, and black labels with small digits indicate unclassified time points. In methane panels, step graphs have been slightly shifted horizontally to prevent overlap. FAN: free ammonia nitrogen. VFA: volatile fatty acid. Download figure Open in new tab Figure 4. Comparison of regularized logistic and linear regression models, which are based upon functional or taxonomic count data, across studies. Models’ mean loss values, calculated by cross-validation, are plotted against the number of included variables, on a logarithmic scale. As loss function, deviance is used for logistic regression and MSE for linear regression. In logistic regression, smallest models with near-minimal mean loss (per count type, per study) are marked with a vertical bar. In linear regression, standard error bars have been added to models with minimal mean loss. MSE: mean squared error. In the final selection of optimal hierarchy levels, the selected levels of taxonomic count types were in majority of high rank; 7/12 had rank class or higher, but low ranks such as species and genus were also selected. The selected rank was sometimes the same between logistic and linear regression; this was the case in count type GAT in the Lemaigre study and in count type 16S in the Cardona study. Sometimes the selected rank diverged substantially between logistic and linear regression however; in the Ahrens study, the selected rank for logistic regression was phylum while the rank for linear regression was species. Regarding the functional KEGG data type, the module level consistently performed best in logistic regression, while the three studies each had a different hierarchical level performing best in linear regression. The optimal hierarchy levels for all count types are included in Fig. 4 , where their logistic and linear regression performance are compared within studies. Taxonomic count types (16S and GAT) generally had better performance than the functional count types (KEGG and eggNOG), in both logistic regression and linear regression. That is, for the same number of variables, in logistic and linear regression respectively they have similar or lower mean deviance and mean MSE (both of which are loss functions, which are lower for better-fitting models). While model deviance approximated zero in all count types in logistic regression as model size increases, indicating that all count types could perfectly predict disturbance state, this low-deviance stage was reached with fewer variables by taxonomic count types. It should be noted, however, that in linear regression, in the Lemaigre study, KEGG outperformed GAT and performed similarly to 16S. It is also noteworthy that in logistic regression, 16S outperformed GAT in the Lemaigre study, while in contrary, GAT outperformed 16S in the Cardona and Ahrens studies. Similarly, in linear regression, 16S outperformed GAT in the Lemaigre study while GAT outperformed 16S in the Ahrens study. After separate analyses, data from the three independent studies were combined into a ‘combined study’ dataset, and analyzed in another regularized regression, which is shown in Fig. 5 . While 16S data was omitted due to different indexing databases, GAT still outperformed the functional data types in logistic regression, but KEGG had the lowest MSE minimum in linear regression. Download figure Open in new tab Figure 5. Comparison of regularized logistic and linear regression models, which are based upon functional or taxonomic count data, for combined dataset. Models’ mean loss values, calculated by cross-validation, are plotted against the number of included variables, on a logarithmic scale. As loss function, deviance is used for logistic regression and MSE for linear regression. In logistic regression, smallest models with near-minimal mean loss (per count type, per study) are marked with a vertical bar. In linear regression, standard error bars have been added to models with minimal mean loss. MSE: mean squared error. Finally, the features of all minimal loss models were extracted, together with their regression coefficients. The number of shared features between logistic and linear regression analysis for the same feature type and study, and the number of shared features between studies for the same feature type and regression type are included in Supplementary Table 5 and Supplementary Table 6, respectively. Although an in-depth description of all included features is beyond the scope of this study, it is remarkable that in all data types, only few features are shared between selected logistic and linear regularized regression models. Only one taxon identified by GAT is shared by both logistic and linear regression, and only two of many extracted eggNOG features are shared. This was the case even when considering that count types can have different hierarchical levels for different studies. However, several taxa assigned based upon 16S data are shared, and for KEGG features, 5/61 modules were selected by both logistic and linear regression in the combined study. An example is M00085 “Fatty acid elongation in mitochondria”. Shared features could not be straightforwardly detected in the separate studies for the KEGG data, because in that type, extracted features were from different hierarchy levels. Also when comparing features between studies using the same regression algorithm (Supplementary Table 6), only few features are shared. While in logistic regression, for KEGG data, 30/98 features were shared between two or more studies and module M00985 “Sulfide oxidation, sulfide => sulfur” was shared between all studies, this was the only feature type where this was the case. In all other combinations of a regression type and data type, a smaller fraction of features was shared. Discussion Given the interest in including microbial community data in biogas monitoring, we have compared functional and taxonomic parameters for detecting process disturbances in reactors. Our regularized regression analyses ( Fig. 4 and Fig. 5 ) show that models based on taxonomic parameters consistently performed similarly as or better than models based on functional parameters. This is the case in three independent lab-scale studies of ammonia disturbance; both when analyzing their data separately and when analyzing the combined data. These results do not support our hypothesis that functional metagenomics data represent the biogas process better than taxonomic metagenomics data. When comparing our results to findings reported in earlier literature, they do not match the statement by Louca et al .[ 15 ] that functional data should be the “baseline” of future microbial studies (they only claimed this for the context of biogeography however). Even in linear regression, taxonomic models gave a similar or better fit for FAN concentrations, which does not support the expectation that functional models, which are more fine-grained, perform better. On the other hand, our results do match the statistically insignificant difference in classification accuracy between 16S data and metagenomic annotation found by Xu et al .[ 14 ], as well as the findings of Casimiro-Soriguer et al .[ 16 ], who found taxonomic data outperforming functional data in a human microbiome context. To explain the presented results, one could hypothesize that functional databases are even more incomplete than taxonomic databases. The biogas system includes many poorly-described micro-organisms[ 38 ]. For these, the functional annotation of orthologs is likely to be based upon experiments in distantly-related organisms, which could reduce the reliability of these annotations. The result that between taxonomic data types, the metagenomic GAT type did not consistently outperform the 16S type is in line with recent results in human microbiome research, where metagenomic data has been found to yield “similar performances” or has performed only “slightly better”[ 49 ]. This implies that for current environmental microbiological analyses, 16S-based techniques may currently suffice; which in present are to be preferred because of their lower cost. However, the fact that different workflows, including different 16S regions and databases, have been used to annotate 16S data in different studies makes it difficult to draw conclusions here. It should also be noted, that the majority vote over contigs, which is included in the full CAT-workflow, could improve the performance of the GAT type further. This majority vote was excluded here for the sake of comparison to functional data, which always function per gene. Challenge: limited power While our results do not support our hypothesis, they cannot conclusively dismiss it. Even though this study included multiple independent data sets, the number of independent reactors is still too small for a statistical test with sufficient power to compare the results of the functional and taxonomic count types. Because of the same reason, no separate dataset was used to perform an independent test of the fitted glmnet models, and to test their performance. Nonetheless, both separate and combined analyses support a difference, albeit not statistically significant. These findings can be used for consecutive studies of functional and taxonomic metagenomics data in microbiology, particularly in the biogas system. The limited number of independent reactors is also the reason why elastic net was chosen as algorithm for supervised analysis. This approach does not allow for the modeling of non-linear effects, and while this is a limitation, it decreases the risk of overfitting the dataset. Another factor that contributed to the risk of overfitting was the multiple hierarchy levels for each count type, further increasing the number of features (compared to the limited number of samples). A connected observation is that within a study, a non-optimal subtype of a taxonomic dataset can have worse performance than a functional subtype. For example, in the Ahrens study, 16S subtype ‘order’ is outperformed by KEGG optimal subtype ‘module’. A final limitation related to machine learning was that multiple samples were derived from identical reactors, violating the generalized linear model assumption of independent observations. The risk of the conclusions in this study being affected is decreased by the fact that these limitations apply to both types of analyses, functional and taxonomic. The lack of shared features selected by glmnet within count types, both between regression types for the same study and between studies for the same regression type, puts the explainability of the models into question. Furthermore, KEGG is the only count type in which a larger fraction of features was shared, between studies for the same regression type. This could support the notion that while functional metagenomics does not outperform taxonomic metagenomics, the former includes more consistent feature profiles and is therefore more explainable. Confirmation of this observation through follow-up research is needed. Perspectives and implications To follow up on this research the performance of other data types could be assessed. Transcriptomic sequencing data would certainly be of interest, as it might give a more reliable functional perspective than metagenomic data, showing which genes are actively being transcribed, and potentially translated, to functional proteins. While the potential of metatranscriptomics is great, handling RNA is more demanding than both 16S and metagenomics. Implementing RNA-based monitoring in biogas processes would therefore likely be more challenging than implementing DNA-based monitoring. Other possible workflows for generating sequencing count data include assembly-free annotation tools, or 16S-based functional annotation tools. We do not, however, expect such changes in the sequence processing to alter the conclusions of our study. The implications of the functional metagenomics not outperforming taxonomic metagenomics, as well as taxonomic metagenomics not consistently outperforming 16S is that in monitoring of the biogas system, metagenomic sequencing might have little added value compared to amplicon sequencing, which is cheaper. Functional analysis could still be of value, however, when trying to understand the mechanisms of changes in a microbial system. Acknowledgments The authors would like to thank Sébastien Lemaigre, Xavier Goux and Magdalena Calusinska for sharing samples from the Lemaigre study. Dries Boers would like to thank the Swedish Bioinformatics Advisory Program, and his program advisors John Sund and Lokeshwaran Manoharan in particular. He would also like to thank Reza Belaghi for advice on classification algorithms; Stan van Lier for valuable discussions on performance metrics in regularized regression; and Nils Weng, Jonas Ohlsson and Malin Tiefensee for helpful comments on the manuscript. This work was supported by the Swedish University of Agricultural Sciences (SLU) and the Swedish Energy Agency, project no. P2022-00552. High-throughput sequence processing was enabled by resources provided by the National Academic Infrastructure for Supercomputing in Sweden (NAISS), partially funded by the Swedish Research Council through grant agreement no. 2022-06725. Funder Information Declared Swedish University of Agricultural Sciences, https://ror.org/02yy8x990 Swedish Energy Agency, https://ror.org/0359z7n90 , P2022-00552 National Academic Infrastructure for Supercomputing in Sweden Swedish Research Council, https://ror.org/03zttf063 , 2022-06725 References 1. ↵ Farghali M , Osman A , Umetsu K , Rooney DW . Integration of biogas systems into a carbon zero and hydrogen economy: a review . Environ Chem Lett . 2022 ; 20 : 2853 – 2927 . doi: 10.1007/s10311-022-01468-z OpenUrl CrossRef 2. ↵ REPowerEU . Communication from the commission to the european parliament, the european council, the council, the european economic and social committee and the committee of the regions . 2022 . p. 230 . Available: https://eur-lex.europa.eu/legalcontent/EN/ALL/?uri=CELEX%3A52012DC0673 3. ↵ Wu D , Li L , Zhao X , Peng Y , Yang P , Peng X. Anaerobic digestion: A review on process monitoring . Renewable and Sustainable Energy Reviews . 2019 ; 103 : 1 – 12 . doi: 10.1016/j.rser.2018.12.039 OpenUrl CrossRef 4. ↵ Nielsen HB , Angelidaki I. Codigestion of manure and industrial organic waste at centralized biogas plants: process imbalances and limitations . Water Science and Technology . 2008 ; 58 : 1521 – 1528 . doi: 10.2166/wst.2008.507 OpenUrl Abstract / FREE Full Text 5. ↵ Jiang Y , McAdam E , Zhang Y , Heaven S , Banks C , Longhurst P. Ammonia inhibition and toxicity in anaerobic digestion: A critical review . Journal of Water Process Engineering . 2019 ; 32 : 100899 . doi: 10.1016/j.jwpe.2019.100899 OpenUrl CrossRef 6. ↵ Gallert C , Bauer S , Winter J. Effect of ammonia on the anaerobic degradation of protein by a mesophilic and thermophilic biowaste population . Applied Microbiology and Biotechnology . 1998 ; 50 : 495 – 501 . doi: 10.1007/s002530051326 OpenUrl CrossRef PubMed 7. ↵ Rajagopal R , Massé DI , Singh G. A critical review on inhibition of anaerobic digestion process by excess ammonia . Bioresource Technology . 2013 ; 143 : 632 – 641 . doi: 10.1016/j.biortech.2013.06.030 OpenUrl CrossRef PubMed 8. ↵ Ferguson RMW , Coulon F , Villa R. Understanding microbial ecology can help improve biogas production in AD . Science of The Total Environment . 2018 ; 642 : 754 – 763 . doi: 10.1016/j.scitotenv.2018.06.007 OpenUrl CrossRef PubMed 9. ↵ Singh A. Microbiological surveillance of biogas plants : focusing on the acetogenic community . [Doctoral thesis]., Swedish University of Agricultural Sciences (SLU) . 2021 . Available: https://pub.epsilon.slu.se/22699/ 10. ↵ Lemaigre S , Adam G , Gerin PA , Noo A , De Vos B , Klimek D , et al. Potential of multivariate statistical process monitoring based on the biogas composition to detect free ammonia intoxication in anaerobic reactors . Biochemical Engineering Journal . 2018 ; 140 : 17 – 28 . doi: 10.1016/j.bej.2018.08.018 OpenUrl CrossRef 11. ↵ Carballa M , Regueiro L , Lema JM . Microbial management of anaerobic digestion: exploiting the microbiome-functionality nexus . Curr Opin Biotechnol . 2015 ; 33 : 103 – 111 . doi: 10.1016/j.copbio.2015.01.008 OpenUrl CrossRef PubMed 12. ↵ Welch RA , Burland V , Plunkett G , Redford P , Roesch P , Rasko D , et al. Extensive mosaic structure revealed by the complete genome sequence of uropathogenic Escherichia coli . Proceedings of the National Academy of Sciences . 2002 ; 99 : 17020 – 17024 . doi: 10.1073/pnas.252529799 OpenUrl Abstract / FREE Full Text 13. ↵ Martiny JBH , Jones SE , Lennon JT , Martiny AC . Microbiomes in light of traits: A phylogenetic perspective . Science . 2015 ; 350 : aac9323 . doi: 10.1126/science.aac9323 OpenUrl Abstract / FREE Full Text 14. ↵ Xu Z , Malmer D , Langille MGI , Way SF , Knight R. Which is more important for classifying microbial communities: who’s there or what they can do? The ISME Journal . 2014 ; 8 : 2357 – 2359 . doi: 10.1038/ismej.2014.157 OpenUrl CrossRef PubMed 15. ↵ Louca S , Jacques SMS , Pires APF , Leal JS , Srivastava DS , Parfrey LW , et al. High taxonomic variability despite stable functional structure across microbial communities . Nat Ecol Evol . 2016 ; 1 : 0015 . doi: 10.1038/s41559-016-0015 OpenUrl CrossRef 16. ↵ Casimiro-Soriguer CS , Loucera C , Peña-Chilet M , Dopazo J. Towards a metagenomics machine learning interpretable model for understanding the transition from adenoma to colorectal cancer . Sci Rep . 2022 ; 12 : 450 . doi: 10.1038/s41598-021-04182-y OpenUrl CrossRef PubMed 17. ↵ Hernández Medina R, Kutuzova S , Nielsen KN , Johansen J , Hansen LH , Nielsen M , et al. Machine learning and deep learning applications in microbiome research . ISME Communications . 2022 ; 2 : 98 . doi: 10.1038/s43705-022-00182-9 OpenUrl CrossRef PubMed 18. ↵ Campanaro S , Treu L , Kougias PG , Luo G , Angelidaki I. Metagenomic binning reveals the functional roles of core abundant microorganisms in twelve full-scale biogas plants . Water Res . 2018 ; 140 : 123 – 134 . doi: 10.1016/j.watres.2018.04.043 OpenUrl CrossRef 19. ↵ Fischer MA , Ulbricht A , Neulinger SC , Refai S , Waßmann K , Künzel S , et al. Immediate Effects of Ammonia Shock on Transcription and Composition of a Biogas Reactor Microbiome . Front Microbiol . 2019 ; 10 . doi: 10.3389/fmicb.2019.02064 OpenUrl CrossRef 20. ↵ Lin Q , Li L , De Vrieze J , Li C , Fang X , Li X. Functional conservation of microbial communities determines composition predictability in anaerobic digestion . The ISME Journal . 2023 ; 17 : 1920 – 1930 . doi: 10.1038/s41396-023-01505-x OpenUrl CrossRef PubMed 21. ↵ Wagenmakers E-J , Wetzels R , Borsboom D , van der Maas HLJ , Kievit RA . An Agenda for Purely Confirmatory Research . Perspect Psychol Sci . 2012 ; 7 : 632 – 638 . doi: 10.1177/1745691612463078 OpenUrl CrossRef PubMed 22. ↵ Cardona L , Mazéas L , Chapleur O. Deterministic processes drive the microbial assembly during the recovery of an anaerobic digester after a severe ammonia shock . Bioresource Technology . 2022 ; 347 : 126432 . doi: 10.1016/j.biortech.2021.126432 OpenUrl CrossRef PubMed 23. ↵ Ahrens L , Schnürer A. Unpublished work . 24. ↵ QIAGEN . AllPrep PowerViral DNA/RNA Kit . Available: https://www.qiagen.com/us/products/discovery-and-translational-research/dna-rna-purification/dna-purification/microbial-dna/allprep-powerviral-dna-rna-kit/ 25. ↵ MO BIO . PowerSoil DNA Isolation Kit . 2016 . Available: https://www.qiagen.com/se/resources/download.aspx?id=5c00f8e4-c9f5-4544-94fa-653a5b2a6373&lang=en 26. ↵ MP Biomedicals . FastDNA Spin Kit for Soil DNA Extraction . Available: https://www.mpbio.com/116560000-fastdna-spin-kit-for-soil-samp-cf 27. ↵ Sund J. NBIS-Metagenomics . NBIS -National Bioinformatics Infrastructure Sweden ; 2021 . Available: https://github.com/NBISweden/nbis-meta 28. ↵ Bolger AM , Lohse M , Usadel B. Trimmomatic: a flexible trimmer for Illumina sequence data . Bioinformatics . 2014 ; 30 : 2114 – 2120 . doi: 10.1093/bioinformatics/btu170 OpenUrl CrossRef PubMed Web of Science 29. ↵ Babraham Bioinformatics . FastQC: A Quality Control tool for High Throughput Sequence Data . 2010 . Available: http://www.bioinformatics.babraham.ac.uk/projects/fastqc/ 30. ↵ Ewels P , Magnusson M , Lundin S , Käller M. MultiQC: summarize analysis results for multiple tools and samples in a single report . Bioinformatics . 2016 ; 32 : 3047 – 3048 . doi: 10.1093/bioinformatics/btw354 OpenUrl CrossRef PubMed 31. ↵ Li D , Liu C-M , Luo R , Sadakane K , Lam T-W. MEGAHIT: an ultra-fast single-node solution for large and complex metagenomics assembly via succinct de Bruijn graph . Bioinformatics . 2015 ; 31 : 1674 – 1676 . doi: 10.1093/bioinformatics/btv033 OpenUrl CrossRef PubMed 32. ↵ Langmead B , Salzberg SL . Fast gapped-read alignment with Bowtie 2 . Nat Methods . 2012 ; 9 : 357 – 359 . doi: 10.1038/nmeth.1923 OpenUrl CrossRef PubMed Web of Science 33. ↵ Danecek P , Bonfield JK , Liddle J , Marshall J , Ohan V , Pollard MO , et al. Twelve years of SAMtools and BCFtools . Gigascience . 2021 ; 10 : giab008 . doi: 10.1093/gigascience/giab008 OpenUrl CrossRef PubMed 34. ↵ Hyatt D , Chen G-L , LoCascio PF , Land ML , Larimer FW , Hauser LJ . Prodigal: prokaryotic gene recognition and translation initiation site identification . BMC Bioinformatics . 2010 ; 11 : 119 . doi: 10.1186/1471-2105-11-119 OpenUrl CrossRef PubMed 35. ↵ Cantalapiedra CP , Hernández-Plaza A , Letunic I , Bork P , Huerta-Cepas J. eggNOG-mapper v2: Functional Annotation, Orthology Assignments, and Domain Prediction at the Metagenomic Scale . Molecular Biology and Evolution . 2021 ; 38 : 5825 – 5829 . doi: 10.1093/molbev/msab293 OpenUrl CrossRef PubMed 36. ↵ Huerta-Cepas J , Szklarczyk D , Heller D , Forslund SK , Cook H , Mende DR , et al. eggNOG 5.0: a hierarchical, functionally and phylogenetically annotated orthology resource based on 5090 organisms and 2502 viruses . Nucleic Acids Research . 2019 ; 47 : 6 . doi: 10.1093/nar/gky1085 OpenUrl CrossRef 37. ↵ Buchfink B , Reuter K , Drost H-G. Sensitive protein alignments at tree-of-life scale using DIAMOND . Nat Methods . 2021 ; 18 : 366 – 368 . doi: 10.1038/s41592-021-01101-x OpenUrl CrossRef PubMed 38. ↵ Campanaro S , Treu L , Rodriguez-R LM , Kovalovszki A , Ziels RM , Maus I , et al. New insights from the biogas microbiome by comprehensive genome-resolved metagenomics of nearly 1600 species originating from multiple anaerobic digesters . Biotechnology for Biofuels . 2020 ; 13 . doi: 10.1186/s13068-020-01679-y OpenUrl CrossRef PubMed 39. ↵ Kyoto Encyclopedia of Genes and Genomes . KEGG orthology database . 17 Jan 2024 . Available: https://www.genome.jp/kegg/ko.html 40. ↵ von Meijenfeldt FAB , Arkhipova K , Cambuy DD , Coutinho FH , Dutilh BE . Robust taxonomic classification of uncharted microbial sequences and bins with CAT and BAT . Genome Biology . 2019 ; 20 : 217 . doi: 10.1186/s13059-019-1817-x OpenUrl CrossRef PubMed 41. ↵ National Center for Biotechnology Information . BLAST databases . 7 Jan 2021 . Available: https://ftp.ncbi.nlm.nih.gov/blast/db/FASTA/ 42. ↵ R Core Team . R: A Language and Environment for Statistical Computing . Vienna, Austria : R Foundation for Statistical Computing ; 2024 . Available: https://www.R-project.org/ 43. ↵ Wickham H , Averick M , Bryan J , Chang W , McGowan LD , François R , et al. Welcome to the tidyverse . Journal of Open Source Software . 2019 ; 4 : 1686 . doi: 10.21105/joss.01686 OpenUrl CrossRef PubMed 44. ↵ Oksanen J , Simpson GL , Blanchet FG , Kindt R , Legendre P , Minchin PR , et al. vegan: Community Ecology Package . 2022 . Available: https://CRAN.R-project.org/package=vegan 45. ↵ Gloor GB , Macklaim JM , Pawlowsky-Glahn V , Egozcue JJ . Microbiome Datasets Are Compositional: And This Is Not Optional . Front Microbiol . 2017 8 . doi: 10.3389/fmicb.2017.02224 OpenUrl CrossRef PubMed 46. ↵ Friedman JH , Hastie T , Tibshirani R. Regularization Paths for Generalized Linear Models via Coordinate Descent . Journal of Statistical Software . 2010 ; 33 : 1 – 22 . doi: 10.18637/jss.v033.i01 OpenUrl CrossRef PubMed 47. ↵ Zou H , Hastie T. Regularization and Variable Selection Via the Elastic Net . Journal of the Royal Statistical Society Series B: Statistical Methodology . 2005 ; 67 : 301 – 320 . doi: 10.1111/j.1467-9868.2005.00503.x OpenUrl CrossRef PubMed Web of Science 48. ↵ Mansournia MA , Geroldinger A , Greenland S , Heinze G. Separation in Logistic Regression: Causes, Consequences, and Control . American Journal of Epidemiology . 2018 ; 187 : 864 – 870 . doi: 10.1093/aje/kwx299 OpenUrl CrossRef PubMed 49. ↵ Bars-Cortina D , Ramon E , Rius-Sansalvador B , Guinó E , Garcia-Serrano A , Mach N , et al. Comparison between 16S rRNA and shotgun sequencing in colorectal cancer, advanced colorectal lesions, and healthy human gut microbiota . BMC Genomics . 2024 ; 25 : 730 . doi: 10.1186/s12864-024-10621-7 OpenUrl CrossRef PubMed View the discussion thread. Back to top Previous Next Posted June 10, 2025. Download PDF Supplementary Material Email Thank you for your interest in spreading the word about bioRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following Comparing the performance of functional versus taxonomic metagenomics for detecting ammonia disturbances in the biogas system Message Subject (Your Name) has forwarded a page to you from bioRxiv Message Body (Your Name) thought you would like to see this page from the bioRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share Comparing the performance of functional versus taxonomic metagenomics for detecting ammonia disturbances in the biogas system Dries Boers , Olivier Chapleur , Anders F Andersson , Anna Schnürer bioRxiv 2025.06.10.658511; doi: https://doi.org/10.1101/2025.06.10.658511 Share This Article: Copy Citation Tools Comparing the performance of functional versus taxonomic metagenomics for detecting ammonia disturbances in the biogas system Dries Boers , Olivier Chapleur , Anders F Andersson , Anna Schnürer bioRxiv 2025.06.10.658511; doi: https://doi.org/10.1101/2025.06.10.658511 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Microbiology Subject Areas All Articles Animal Behavior and Cognition (7635) Biochemistry (17691) Bioengineering (13892) Bioinformatics (41937) Biophysics (21452) Cancer Biology (18588) Cell Biology (25504) Clinical Trials (138) Developmental Biology (13378) Ecology (19899) Epidemiology (2067) Evolutionary Biology (24320) Genetics (15609) Genomics (22506) Immunology (17736) Microbiology (40394) Molecular Biology (17181) Neuroscience (88605) Paleontology (666) Pathology (2832) Pharmacology and Toxicology (4824) Physiology (7641) Plant Biology (15156) Scientific Communication and Education (2045) Synthetic Biology (4294) Systems Biology (9825) Zoology (2271)

Text is read by the "Ask this paper" AI Q&A widget below. Extraction quality varies by source — PMC NXML preserves structure cleanly, OA-HTML may include some navigation residue, and OA-PDF can have broken hyphenation. The publisher copy (via DOI) is the canonical version.

My notes (saved in your browser only)

Ask this paper AI returns verbatim quotes from the full text · source: preprint-html

Answers must be backed by verbatim quotes from this paper's full text. Hallucinated quotes are dropped automatically; if no verbatim passage answers the question, we say so. How this works

Citation neighborhood (no data yet)

We don't have any in-corpus citations linked to this paper yet. This is a recent paper (2025) — citers typically take a year or two to land, and the OpenAlex reference graph may still be filling in.

Source provenance

europepmc
last seen: 2026-05-20T01:45:00.602351+00:00
unpaywall
last seen: 2026-06-05T02:00:03.366016+00:00
License: CC-BY-4.0