Full text
71,627 characters
· extracted from
preprint-html
· click to expand
serojump: A Bayesian tool for inferring infection timing and antibody kinetics from longitudinal serological data | medRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-P4HH5NV'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search serojump : A Bayesian tool for inferring infection timing and antibody kinetics from longitudinal serological data View ORCID Profile David Hodgson , View ORCID Profile James Hay , Sheikh Jarju , Dawda Jobe , Rhys Wenlock , View ORCID Profile Thushan I. de Silva , Adam J. Kucharski doi: https://doi.org/10.1101/2025.03.04.25323335 David Hodgson 1 Centre for Mathematical Modelling of Infectious Diseases, London School of Hygiene and Tropical Medicine Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for David Hodgson For correspondence: david.hodgson{at}lshtm.ac.uk James Hay 2 Pandemic Sciences Institute, Nuffield Department of Medicine, University of Oxford Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for James Hay Sheikh Jarju 3 Vaccines and Immunity Theme, MRC Unit The Gambia at the London School of Hygiene and Tropical Medicine Find this author on Google Scholar Find this author on PubMed Search for this author on this site Dawda Jobe 3 Vaccines and Immunity Theme, MRC Unit The Gambia at the London School of Hygiene and Tropical Medicine Find this author on Google Scholar Find this author on PubMed Search for this author on this site Rhys Wenlock 3 Vaccines and Immunity Theme, MRC Unit The Gambia at the London School of Hygiene and Tropical Medicine Find this author on Google Scholar Find this author on PubMed Search for this author on this site Thushan I. de Silva 3 Vaccines and Immunity Theme, MRC Unit The Gambia at the London School of Hygiene and Tropical Medicine 4 Division of Clinical Medicine, School of Medicine and Population Health, University of Sheffield 5 Florey Institute of Infection and Sheffield NIHR Biomedical Research Centre, University of Sheffield Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Thushan I. de Silva Adam J. Kucharski 1 Centre for Mathematical Modelling of Infectious Diseases, London School of Hygiene and Tropical Medicine Find this author on Google Scholar Find this author on PubMed Search for this author on this site Abstract Full Text Info/History Metrics Supplementary material Data/Code Preview PDF ABSTRACT Understanding acute infectious disease dynamics at individual and population levels is critical for informing public health preparedness and response. Serological assays, which measure a range of biomarkers relating to humoral immunity, can provide a valuable window into immune responses generated by past infections and vaccinations. However, traditional methods for interpreting serological data, such as binary seropositivity and seroconversion thresholds, often rely on heuristics that fail to account for individual variability in antibody kinetics and timing of infection, potentially leading to biased estimates of infection rates and post-exposure immune responses. To address these limitations, we developed serojump , a novel probabilistic framework and software package that uses individual-level serological data to infer infection status, timing, and subsequent antibody kinetics. We validated serojump using simulated serological data and real-world SARS-CoV-2 datasets from The Gambia. In simulation studies, the model accurately recovered individual infection status, population-level antibody kinetics, and the relationship between biomarkers and immunity against infection, demonstrating robustness under observational noise. Benchmarking against standard serological heuristics in real-world data revealed that serojump achieves higher sensitivity in identifying infections, outperforming static threshold-based methods and precision in inferred infection timing. Application of serojump to longitudinal SARS-CoV-2 serological data taken during the Delta wave provided additional insights into i) missed infections based on sub-threshold rises in antibody level and ii) antibody responses to multiple biomarkers post-vaccination and infection. Our findings highlight the utility of serojump as a pathogen-agnostic, flexible tool for serological inference, enabling deeper insights into infection dynamics, immune responses, and correlates of protection. The open-source framework offers researchers a platform for extracting information from serological datasets, with potential applications across various infectious diseases and study designs. AUTHOR SUMMARY Tracking how infections spread and how our immune systems respond to them is essential for improving public health. One way to study this is by analysing blood samples to measure antibody levels, which can help estimate who has been infected, when they were infected, and how their immune response has developed over time. However, traditional methods for interpreting these antibody levels often use simple cutoffs that do not account for how much people’s immune responses can vary, which can lead to inaccurate results. To solve this, we created a new tool in R called serojump . This tool uses advanced statistical methods to analyse changes in antibody levels over time, helping to more accurately identify who has been infected, when they were infected, and what happens to their antibodies afterwards. In our tests, serojump was better at detecting infections than traditional methods and provided more detailed insights about how people’s immune systems responded to the virus and vaccines. It also revealed infections that were missed by standard testing methods. Our tool is flexible and can be used for many different diseases. By helping researchers and public health workers better understand infection patterns and immune responses, serojump can support efforts to control the spread of diseases and develop more effective treatments and vaccines. 1. INTRODUCTION Serological samples can be analysed to detect the presence of biomarkers, such as antibodies, made in response to an infection long after the infection has cleared.[ 1 ] By combining validated biomarker assays with appropriate statistical inference techniques, it is increasingly possible to estimate incidence rates, antibody kinetics, and population-level susceptibility without ongoing syndromic surveillance.[ 2 ] Different types of serological assays contribute to such analyses, including morphological assays such as ELISA, which measure antibody concentrations, and functional assays such as neutralisation tests, which measure the ability of antibodies to inhibit infection in vivo. Analysing serological samples is important because syndromic surveillance systems typically only capture cases of disease, representing the upper parts of the reporting pyramid.[ 3 ] However, effective control and prediction of pathogen spread require understanding the full spectrum of the underlying epidemiology, including both symptomatic and asymptomatic infections, as asymptomatic or pauci-symptomatic individuals may also contribute to transmission dynamics and population immunity. Therefore, analysing serological samples can enable researchers and healthcare professionals to infer crucial information about the epidemiology of a pathogen at the individual and population level, which case-based surveillance systems may otherwise miss.[ 4 ] Such analysis can in turn help the understanding of the immune system’s ability to combat various pathogens, aid in developing new targeted intervention programmes and provide insights into patterns and drivers of the transmission dynamics of infectious diseases. On the individual level, past infection with a specific pathogen has traditionally been inferred from measured antibodies using either i) an antibody threshold level (i.e. seropositivity) or ii) a threshold fold-rise between a pair of samples (i.e. seroconversion).[ 3 ] Considerable research has focused on understanding how seropositivity and seroconversion rates change according to controlled host factors, such as age, geography, living conditions, sexual behaviour, etc.[ 5 – 7 ] On the population level, serological samples which are representative of a population (e.g. cross-sectional samples) can be used to estimate the prevalence of infectious diseases (seroprevalence) and determine how seroprevalence changes over time according to host factors.[ 8 – 10 ] Estimation of infection through analysis of seropositivity or seroconversion requires deriving an accurate absolute or relative threshold value, and these are often determined by rule-of-thumb heuristics (e.g. for influenza: 4-fold-rise for conversion, titre of 1:40 HAI for seropositivity).[ 11 ] However, antibody responses vary greatly between individuals for many pathogens. Therefore, relying on these pre-determined heuristics to determine infections in serology studies can lead to both false positives and false negatives in inferred infection status, leading to biased estimates of prevalence.[ 8 , 12 , 13 ]) Consequently, there has been a growth in new analytical methods to make better use of quantitative measurements derived from serological samples to inform infectious disease epidemiology and public health policy, a field of research termed ’serodynamics’.[ 2 ] In particular, efforts have been made to probabilistically infer individuals’ infection status and timing in a serological cohort by analysing longitudinal changes in serological titre.[ 14 – 16 ] By modelling the expected antibody level over time following infection or vaccination, single or multiple measurements of an individual’s antibody level can be used to back-calculate the presence of and likely timing of infection. These methods of inferring infection using an individual’s changes in antibody levels are called “time-since-infection” or “titre-based” methods. However, the complexity of modelling these uncertain parameters—such as antibody kinetics, individual infection status, and the timing of infection—requires sophisticated statistical methods. Standard Bayesian inference techniques, such as the Metropolis-Hasting algorithm, cannot sample over the high-dimensional, transdimensional space these models create, particularly when infection status and timing are being inferred. One promising approach is reversible jump MCMC, a transdimensional sampler that can explore infection status and timing on a continuous scale.[ 17 ] This method has already provided valuable insights in specific studies, revealing patterns of infection and immunity for dengue and influenza that would otherwise remain hidden.[ 15 , 16 ] However, its application has been limited to narrowly defined problems due to the complexity of its underlying statistical methods and the need for problem-specific customisation of the sampler itself. This restriction prevents the broader adoption of reversible jump MCMC in epidemiological research despite its potential to infer accurate estimates of infection risk, correlates of protection, and antibody kinetics using only serological data. Therefore, our study introduces serojump , a general-purpose statistical framework that applies reversible jump MCMC to infer infection status, timing, and antibody kinetics from serological data. By leveraging individual-level longitudinal changes in biomarker values, serojump provides a probabilistic approach to estimating epidemiological parameters, addressing the limitations of static serological heuristics. We validate this framework through simulation studies and real-world SARS-CoV-2 serological data from The Gambia, demonstrating its ability to recover infection histories, improve sensitivity in infection detection compared to traditional methods, and infer correlates of protection. By making these advanced inference techniques accessible via an R package, serojump enhances the utility of serological data for infectious disease surveillance and public health decision-making. 2. METHODS 2.1. Overview of inference framework Using data on individual-level temporal changes in serological biomarker values, serojump can infer several key quantities within a single probabilistic framework: i) the mean population-level kinetics of the biomarker; ii) the individual-level probability of infection during a specific period; and iii) the distribution of timing of this infection. Defining as the measured value at time,, to biomarker, for individual ; as the latent binary vector representing the infection status of each individual (1: infected, 0: not infected), and = {τ i } is the infection time for those infected (where the length of the vector is equal to the number of infected individuals) we define the likelihood of this algorithm as in Menezes et al:[ 18 ]: Where, is the observational model, describing the likelihood of the serological data,, given the model predicted latent titre values,, (defined by the function,) and represents the model parameters being estimated. The posterior distribution is therefore defined by: Where is the prior distribution for the timing of infection for individual i , is the prior distribution on the infection status binary vector, and is the prior distribution on the parameters describing the observational model and antibody kinetics model. To sample from EQUATION 2, we used a reversible-jump MCMC algorithm, which allows for efficient exploration over the fixed dimensions of and the dynamically changing dimensions of The full algorithm and its derivation are given in Supplementary Methods section 1.1–1.3 . 2.1.1 Antibody kinetics model In the most general case, the antibody kinetics model, for an individual i , we define k immunological stimuli, {e 1 , e 2 , …, e k-1 , e k } which occur at times {t e1, t e2 , … t ek-1, t ek }, such that t ei < t ei+1 for all i. These could represent stimuli such as exposure to an infection or vaccine. If the functions f b e1 , f b e2 , …, f b ek-1 , and f b ek represent the antibody kinetics following immunological stimulus e i for biomarker b , then, given a serological sample at time t ek < t* < t ek+1, we estimate the biomarker value (e.g. log neutralising titre) at this time A i, t b in the model as: That is, the antibody titre is governed by an individual’s most recent exposure and the function for this response is conditional on the time since this exposure, the biomarker level at that time, and any exposure-specific parameters. Given that our analysis focuses on inferring infections, we assume that for a subset of individuals, the immunological stimuli, e j , results in an infection, and the timing of this infection r i is t ej . We refer to this framework as an antibody kinetics model as, in this study, we are measuring titre values to antibody levels, but note the same method can be applied to any measured immunological biomarker (e.g. memory B cell concentrations). 2.1.2 Observational model The observation model defines the likelihood function that links the model-estimated biomarker value at the sampled time to the measured value at that time. Typically, serological measurements are assumed to follow a normal distribution: . However, because all assays have upper and lower limits of detection, a censored normal distribution is often used to account for interval censoring. Additionally, many serological data points are discrete due to measurement constraints—such as dilution series in assays (e.g., 1:10, 1:20, 1:40, etc.)—where the true biomarker level is continuous, but observations are recorded in discrete steps. In such cases, a likelihood function that explicitly models interval-censored or discretized normal distributions is required to accurately capture the measurement process.[ 19 ] 2.1.3 Infection model The prior for the infection time,, for each individual can be estimated on the population-level from existing serological surveys or surveillance data to identify likely periods of infection. For example, surveillance data can still indicate the distribution of infection burden over time, if not the true cumulative number of infections. A key feature of our framework is that can be specified to reflect a shared force of infection (FOI) across individuals, ensuring that infection times are conditionally dependent given the population-level FOI. This avoids the naive assumption that values are fully independent for each individual. If this prior is not specified in the model, or there is no known data to infer the epidemic curve, the model will assume a uniform risk of infection over the study period ( Supplementary Methods section 1.4) . 2.1.4 Prior distributions The prior distribution on the parameters describing the observational model and antibody kinetics model, can be defined by the user and are assumed to be independent. The prior distribution on the infection status binary vector, as with similar inference problems,[ 19 ] has an implicit prior resulting from the definition of the reversible jump algorithm. This is Bin(0.5, N), or 50% of the population infected on average. To remove this implicit prior and instead assign a uniform distribution to the total number of infected individuals during the epidemic, we choose the combinatoric prior: If prior knowledge on the number of missed infected individuals is known (e.g. from random community testing regardless of symptoms), these can be added to this prior ( Supplementary Methods section 1.4–1.5). 2.1.5. Post-processing of posterior distributions and correlates of protection After fitting the serojump algorithm, we obtain posterior distributions for the set (τ̂, Ẑ , θ̂ ). We calculate the posterior distribution of the post-exposure antibody trajectories for exposure e and biomarker b, f b e ( t , θ̂ ). To assess the posterior distribution of the number of infected individuals across the population, we calculate for each sample of the posterior distribution.A key estimate in serological studies is the relationship between a biomarker (e.g., antibody titre) and infection risk, where the biomarker is referred to as a correlate of protection (COP) . A COP represents either a direct mediator of immunity or a proxy for immune protection. To estimate this relationship from serojump , we fit a logistic curve with parameters k (gradient) and x 0 (mid-point) to describe how reference biomarker values relate to the probability of infection, given by: From this, we define the absolute correlate of protection (aCOP) as the fitted probability curve, representing the direct relationship between biomarker levels and protection from infection. It can be derived from the fitted R(x) as: The relative correlate of protection (rCOP) , is a rescaled version of the aCOP that normalizes protection relative to a reference titre level (in this cas the lower limit of detection. It can also be derived from fitted R(x) as These estimates allow us to quantify the protective effect of immune markers and assess how biomarker levels modulate infection risk across a population ( Supplementary Methods section 1.6–7 ). 2.2 Application to simulated data 2.2.1 Description of the simulated data To test whether it is possible to correctly recover epidemiological and immunological dynamics from common serological cohort structures, we first simulate a serological dataset using the serosim R package[ 18 ] to test the serojump framework. We simulate continuous epidemic serosurveillence (CES) cohort data, which represents a serostudy in which individuals are followed over an epidemic wave and sampled at multiple random time points throughout. The simulated data includes N = 200 individuals with serological samples taken within the first seven days of the study’s starting and a sample within the last seven days of the study’s ending. These individuals also had three samples taken randomly throughout the study during an 120-day epidemic wave. We model the epidemic as a stochastic transmission process, where individuals can be either susceptible, exposed, or infected. The probability of exposure is set at 60% per individual, meaning that individuals coming into contact with an infectious person have this probability of becoming exposed, assuming no prior immunity. Among those exposed, the probability of progressing to infection is 30%, meaning that some individuals remain uninfected despite exposure. This corresponds to a scenario where a pathogen spreads within a population that starts with partial immunity. Over the study timeframe we assume that individuals can have a maximum of one exposure. To model a symmetric epidemic peak, we simulate the exposure time from a normal distribution, N (60, 20) days. To assess the impact of biomarker levels on infection risk, we simulate two different relationships between the measured biomarker and the probability of infection given exposure: (1) No COP model (Model A), where infection occurs with a fixed probability of 50% regardless of biomarker titre, representing a non-titre-dependent correlate of protection ; and (2) With COP model (Model B), where infection probability follows a logistic function of biomarker titre, a commonly used correlate of protection model [ 20 , 21 ]. Explicitly, the simulated probability of infection, given a titre value, x, is given by f(x) = 1 / 1 +exp(−(−4 +2x)). Following infection, the antibody kinetics are assumed to follow a linear rise to a peak at 14 days, followed by an exponential decay to a set-point value.[ 22 ] The formula for this biphasic trajectory is given by The simulated values we use are a = 1.5, b = 0.3, c = 1 and t , is the number of days post-infection. These values are calibrated to produce a biphasic response where antibodies rise sharply during the first 14 days post-infection (representing immune activation), followed by a gradual ex ponential decline after peak response, stabilising at a long-term set point. This pattern is commonly observed in infections such as influenza and SARS-CoV-2. If an individual is not infected over the period, their titre remains unchanged and, therefore, equal to their start titre throughout. We assume all individuals have the same antibody kinetic response following infection but we add observational noise into our model, assuming a normal distribution with standard deviation σ. In the base case, we assume σ = 0.1. We tested the robustness of the simulation recovery to noise by fitting the serojump algorithm to simulated data with increasing observational error. Specifically, we considered standard deviations in the normally distribution observation error for 11 values (0.01, 0.05, 0.1, 0.15, .., 0.45, 0.5), and evaluated the capacity for serojump to recover i) the infection status of individuals, ii) the timing of the epidemic and iii) the recovered population-level antibody kinetics. 2.2.2 Model specification for serojump We assume each individual’s serological profile is characterised by the IgG biomarker (b={IgG}). Two immunological events are considered: e pre , the pre-infection state and e inf , the infected state. The structural assumptions on the kinetics of the pre-infection and infection state are; for the pre-infection state, we assume the prior on the antibody levels Y i are modeled as a linear wane with a waning rate with the uniform distribution with prior w ∼ U(0,0.1). For the post-infection dynamics, we assume that antibody-kinetics boost according to the functional definition defined as ( Equation 5 ) with the priors, a ∼ N (2, 2), b ∼ N (0.3, 0.05), c ∼ U (0, 4). For the observational model, we assume that antibody measurements follow the normal distribution probability density function [ F N ( X |µ, σ)]; with the prior a ∼ U(0, 4). For the prior on the risk of infection over time, we assume this is uniform across the whole time period P exp (τ i ) = U (1, 120) for all i . Finally, the distribution for the number of infection events, P ( Z ), is constrained by a combinatorial factor reflecting the arrangement of events across the population as described in Equation 4 . A summary for the method specification for serojump can be found in the Supplementary Methods Section 2 . 2.3 Application to empirical data 2.3.1 Description of the real-world serological data on SARS-CoV-2 from The Gambia To test serojump on real-world serological data, we used longitudinal serological data collected before and after the Delta wave [ 23 ] in a household cohort study of 349 individuals in The Gambia. Quantitative ancestral anti-spike and nucleocapsid (NCP) IgG data were generated using previously validated ELISA assays,[ 10 ] calibrated to the WHO International Standard for anti-SARS-CoV-2 immunoglobulin (cat no NIBSC 20/136). In addition to serological data, SARS-CoV-2 PCR testing was conducted weekly regardless of symptoms and additionally for those who presented with influenza-like illness symptoms during this period. In addition to PCR-positive diagnoses, dates of SARS-CoV-2 vaccination and PCR-positive pre-Delta infections were also known. At the start of the study, which was just prior to the Delta variant circulating, 56.7% of people were seropositive for SARS-CoV-2.[ 23 ] Over the study period, the Delta variant predominantly circulated with 99 PCR-confirmed cases, with an additional 21 PCR confirmed SARS-CoV-2 infections with pre-Delta variants, and 49 individuals who received a dose of vaccine. A description of the timing of the sample collection, and infections for each individual is given in Supplementary Methods Section 3 . 2.3.2 Model specification for serojump Each individual’s serological profile is characterized by two biomarkers: IgG to SARS-CoV-2 ancestral spike and NCP protein (b={spike, NCP}). We model four immunological events: e pre , the pre-infection state, e delta , the Delta-infected state, e pre-delta, the pre-Delta infection state, and e vax the vaccination state. For the preinfection state, we assume the prior on the antibody levels Y i for biomarker b are modeled as a linear wane: with a waning rate, w , with a prior w ∼ U(0, 0.1). This parameter is estimated separately for each biomarker.. For post-infection (pre-Delta and Delta) and vaccination, we assume that antibody kinetics follow a boosting function as described in Teunis et al.[ 24 ] () and that the level of boosting is scaled based on the l og-transformed titre values at exposure (such that, where and Where, v = 0.001, and the priors governing the boosting response are given by: r ∼ U(0, 1), y 1 ∼ U(0, 6), t 1 ∼ U(7, 21), and ∼U(0, 1). Thus, in the posterior distribution, each immunological event (pre-Delta infection, Delta infection, vaccination) and each biomarker has distinct parameter values describing its kinetics, but they share the same prior distributions.. For the observational model, we assume that antibody measurements follow the normal distribution probability density ; with prior a ∼ Exponential(1). We choose an empirical distribution based on the PCR positive data for the prior on the risk of infection over time. Finally, the distribution of a number of infection events, P ( Z ), is constrained by a combinatorial factor reflecting the arrangement of events across the population as described in Equation 4 . A summary for the method specification for serojump can be found in the Supplementary Methods Section 3 . 2.3.3 Benchmarking against standard serological heuristics We evaluate how well serojump and other serological approaches can infer true infection status (as defined by PCR-positivity) using serological data alone, without incorporating PCR results into the inference process. Specifically, we compare five different infection classification strategies based on serology: serojump, a four-fold rise in spike or NCP, and seropositivity thresholds for spike and NCP (IC50 values of 6.39 and 3.00) [ 23 ]. To assess the performance of these classification approaches, we compute and compare their sensitivity in detecting PCR-confirmed infections. 2.4. Implementation The code is written in R (v.4.4.1)[ 25 ] and Rcpp (v1.0.14)[ 26 ], open source and packaged at https://github.com/seroanalytics/serojump . The package is flexible, allowing the user to input their functional forms and associated prior distributions for the antibody kinetics model (f b ) for each immunological stimulus and biomarker, or the user can select built-in functions from previous papers such as those mentioned in this study. Further, the user can specify their functional form for the observational model likelihood [P obs ( b |A b )] and the prior distributions on the arguments. Finally, the user can customise the prior on the risk of infection over the time period, P exp (r i ), and the prior on the distribution of infection events P(Z). The inputs for serojump to run the model on the simulated data and empirical data are summarised in T ables S1–4 . The package serojump has vignettes to reproduce this study, which can be found at https://seroanalytics.org/serojump/articles/ . The assessment of convergence statistics is summarised in Supplementary Methods section 1.8 . 3. RESULTS 3.1 CASE 1: SIMULATED DATA 3.1.1. Simulation recovery for the base case simulated dataset For the default simulated dataset with low observational noise (Cf = 0.1), the individual-level antibody kinetics show good agreement between simulated and recovered kinetics under the With COP and No COP simulated data ( Figure 1A–B ) with the observed data points (grey dots) close to the median posterior distributions of the model-fitted individual-level antibody trajectories (red lines). We recover the infection status of each individual with 100% sensitivity and specificity, and the differences between simulated infection days and model-predicted infection days ( Figure 1C–D ) are small (within 14 days) for almost all individuals. This results in a close alignment between the simulated and the recovered epidemic curves ( Figures 1E–F ). These observations suggest that for the default simulated dataset, both the With COP and No COP data are well recovered, as evidenced by closer alignment of posterior medians to observed data, with minor errors in infection time predictions and an accurate reconstruction of the infection time distribution. Download figure Open in new tab Figure 1: Comparison of simulation recovery for With COP and No COP models with observational error (0.1). (A , B) Individual-level antibody kinetics for a subset of individuals, showing observed data points (grey dots), and individual-level posterior medians (red lines) for With COP data and No COP data. (C, D) Model error in recovering infection times, represented as the difference between simulated and model-predicted infection days for infected individuals With COP data and No COP data. (E, F) Distribution of infection times, comparing simulated infection times (grey bars) with recovered infection times (red bars). 3.1.2. Stability of simulation recovery for increasing observational noise We assess predictive performance by measuring the Continuous Ranked Probability Score (CRPS), a metric that evaluates the accuracy of probabilistic predictions by comparing the predicted cumulative distribution function to the observed value. We find the functional form of the antibody kinetics for both No COP and With COP models is well recovered (CPRS < 0.35) across all levels of observational noise; however, decreasing in accuracy as observational noise increases ( Figure 2A ). In contrast, the recovery of the epidemic curve is adequate at low levels of uncertainty but quickly becomes inaccurate (CPRS > 0.7) at a moderate level of noise ( Figure 2B ). This suggests that for datasets with a high degree of observational error, the recovery of the epidemic timing may be inaccurate when a uniform prior on infection timing is used in serojump . Download figure Open in new tab Figure 2: Evaluation of model performance under varying levels of observational uncertainty. (A) Accuracy in recovering antibody kinetics over various simulated observational errors, measured as the mean CRPS across all individuals of the observational model (B) Accuracy in recovering the epidemic curve over various simulated observational errors, assessed using the mean CPRS across all infection times (C, D) ROC curves for infection status predictions under simulated data with COP ( C ) and without COP ( D ) under different observational error (colour) (E) F1-scores for recovering infection status across different observational errors and model types (No COP and With COP). (F, G) Posterior means of the COP curves across various observational uncertainty values (blue colours) compared to the simulated function (dashed red line) (H) Accuracy in recovering the functional form for the COP, measured by Mean CRPS, across various levels of observational uncertainty for both model types (With COP and No NOP) For recovery infection status, we find the model perfectly recovers the infection status of each individual (F1-score = 1) up until an observational error of 0.15, after which there is a decrease in F1-score as observational error increases, suggesting the sensitivity and specificity of the recovery of infection status is decreasing ( Figure 2C–E ). This decrease is driven by the increase in the false positive rate, with the true positive rate remaining consistent across all observational noise, suggesting that the model correctly identifies all infections but may incorrectly detect infections at high levels of noise. For recovery of the COP functional form, we find that when the observational noise is less than 0.3, it is well recovered for both the With COP and No COP datasets (CRPS < 0.05) ( Figure 2F –H ) . However, above this observational error, the accuracy of the recovery of the functional form of the COP begins to decrease rapidly. 3.2 CASE STUDY 2: SEROLOGICAL INFERENCE FOR SARS-CoV-2 IN THE GAMBIA DURING THE DELTA WAVE 3.2.1. Testing sensitivity of serological detection methods We evaluate the sensitivity of various serological detection heuristics in identifying infection status. Using each of the five infection classification strategies based on serology ( serojump , a four-fold rise in spike or NCP, and seropositivity thresholds for spike and NCP), we classified each of the PCR-confirmed individuals by their derived infection status and then calculated the sensitivity of each method ( Figure 3 ). The serojump method demonstrates the highest sensitivity, outperforming the other four heuristics. The four-fold spike rise achieves the second-highest sensitivity, closely followed by the four-fold NCP rise. Both seropositive threshold methods (spike and NCP) perform less well, highlighting the limitations of using simple thresholds for infection detection. These observations highlight that leveraging dynamic changes in antibody kinetics can outperform static thresholds for serological detection of infection, with the serojump method demonstrating superior sensitivity, indicating its potential as a preferred approach for infection inference. Download figure Open in new tab Figure 3. Sensitivity of serological detection heuristics in identifying infection status using serological data only (A) Posterior probabilities of recovery for individuals using five detection methods: (1) Four-fold spike rise, (2) Four-fold NCP rise, (3) Seropositive threshold spike, (4) Seropositive threshold NCP, and (5) serojump . Green points represent individuals inferred as true positives, while red points indicate false negatives. 3.2.2. Combining PCR and serological data to infer epidemiological dynamics of the Delta wave We estimate the dynamics of antibody responses, epidemic patterns, and infection risk in the context of SARS-CoV-2 infection and vaccination during the Delta wave in The Gambia. This uses the same serological dataset described in section 3.1 but includes PCR-confirmed infections, anchoring the infection status and infection time for 99 individuals. We infer antibody kinetics trajectories for the biomarkers, spike and NCP, for individuals infected prior to the Delta variant, during the Delta wave, and following vaccination ( Figure 4A ). We find that infections with Delta and infections with variants before Delta stimulate both spike and NCP responses, with spike responses seeing more sustained responses compared to NCP responses (AUC: 395 for spike and 257 for NCP for Delta infection). The spike responses remain above a four-fold rise up until day 46 for infection with the Delta variant.. Following vaccination, we find robust spike responses but very low NCP responses (AUC: 607 vs 67 for spike and NCP, respectively). These observations are consistent with the spike-based composition of the vaccine used. We also summarise the total number of infections inferred from the serojump framework and compare this to the known PCR-confirmed infections ( Figure 4B , C ). Download figure Open in new tab Figure 4: Antibody kinetics, recovery of infection timings, and correlates of protection from empirical data with PCR information. (A) Fitted antibody kinetic trajectories for individuals with infection prior to the Delta variant (left), infection during the Delta variant wave (center), and following vaccination (right), showing fold-rise in antibody titre over time for NCP (blue) and spike (orange) biomarkers. Shaded regions represent 95% credible intervals. (B) Inferred epidemic wave during the Delta wave, illustrating the temporal distribution of PCR-confirmed cases (black bars) and total inferred infections (pink shaded curve), with shaded uncertainty. (C) The estimated attack rate during the Delta wave, partitioned into PCR-confirmed cases (dark grey) and total inferred infections (pink). Error bars indicate 95% credible intervals. (D) The fitted logistic curve to the infection risk, showing the posterior probability of infection as a function of antibody titre at infection for NCP (blue) and spike (orange) biomarkers, with shaded regions representing 95% credible intervals. (E) The absolute COP from the fitted infection risk curve and (F) shows the relative COP for the same biomarkers. Considering only PCR-confirmed infections, we find that the attack rate is 30%, increasing to 52% when we use serojump , implying analysing changes in serological kinetics finds many infections which are missed through virological surveillance alone. Using our informed prior ensures that the inferred timings of these missed infections agree with the known epidemic wave. Finally we compared the relationship between titre levels at infection and the individual-level posterior probability of infection, serving as a correlate of protection (COP) ( Figure 4D ). For NCP and spike biomarkers, a negative relationship is observed between titre levels at infection and the probability of infection, suggesting that higher antibody titres are associated with reduced infection risk. When we consider fitting the absolute COP after normalising the titre scales, spike has a slightly steeper gradient compared to NCP, suggesting a stronger COP ( Figure 4E ) . In addition, high titre values providing only 60% protection suggest the Ancestral spike and NCP proteins correlate with protection despite being antigenically distant from the infecting variant (Delta), suggesting cross-immunity with the analysed proteins. The relative COP show that having a log spike titre of 3.2 provides 50% more protection against infection compared to those with no measurable titre ( Figure 4F ). 4. DISCUSSION In this study, we present an open-source flexible framework for serological inference, serojump , which uses a reversible-jump MCMC algorithm to probabilistically estimate whether an individual has been infected or not. Specifically, it can infer individual-level infections, antibody kinetics, and population-level dynamics using longitudinal serological data. Our approach uses mechanistic modelling of antibody kinetics to estimate latent infection states and provides an alternative to conventional serological heuristics, which are often based on crude static thresholds. Our results demonstrate that serojump can achieve high sensitivity and specificity in recovering infection status, infection timing, and population-level epidemic patterns from simulated and real-world datasets. The framework effectively captures antibody kinetics post-infection and vaccination, allowing us to differentiate between infections during different epidemic waves and post-vaccination responses. Specifically, the inferred kinetics show that vaccination induces robust spike antibody responses but negligible NCP responses, consistent with the composition of SARS-CoV-2 vaccines and previous observations.[ 27 , 28 ] Our results also show how static heuristics, such as four-fold rises in antibody titres or seropositivity thresholds, underperform compared to the serojump framework. This underscores the need for dynamic, data-driven methods to accurately infer infections, especially in scenarios where virological data are incomplete or unavailable. For example, using real-world data from The Gambia, serojump inferred an attack rate significantly higher than that estimated by PCR-confirmed cases alone, suggesting under-detection by virological sampling methods, potentially due to a relatively short period of peak viral shedding.[ 4 ] Our framework addresses several challenges associated with serological studies, including variability in antibody responses and the need for flexible, biologically informed models. By integrating individual-and population-level data into a unified probabilistic framework, serojump offers a powerful tool for assessing pathogen transmission dynamics, evaluating vaccine efficacy, and identifying correlates of protection. These capabilities are critical for informing public health interventions, especially in resource-limited settings where virological sampling may be sparse. Furthermore, the negative relationship between antibody titres at infection and the individual-level probability of infection observed in our analysis highlights the potential for serojump to identify correlates of protection. These insights can guide the development of immunological benchmarks for vaccine efficacy and inform strategies for boosting population immunity. We highlight that the framework is currently pathogen agnostic and thus is adaptable to multiple pathogens with different serological profiles, highlighting the robustness of the model in handling diverse datasets, which could include varying sample sizes, intervals, and noise levels. Inferring latent infection status and timings from serological data is a highly complex inference challenge. This requires jointly accounting for epidemic dynamics, variation in individual infection timing and status, and complex antibody kinetics that depend on the timing, nature, and antigens of infections. While it would be ideal to have a single framework capable of addressing all these mechanisms simultaneously, such an approach is challenging and demands extensive data. As a result, software tools must be tailored to address different dimensions of this complexity. serosolver is a general-purpose statistical inference framework that uses longitudinal and cross-sectional serological data to determine individual-level infection history, infection timings, and antibody kinetics.[ 19 ] In contrast, serojump focuses on estimating single infection events with precision, both in time and infection status. serojump also provides the ability to estimate the relationship between measured biomarkers and the probability of infection, making it particularly useful for reconstructing epidemic curves over a single epidemic. A key distinction is that serojump uses continuous-time sampling, and offers the user enhanced flexibility in modeling antibody kinetics and observational structures. However, serojump is optimized for short-term dynamics and precise infection timing rather than handling multiple infections over long timeframes or incorporating complex antigenic dynamics, which are key strengths of serosolver . Notably, both tools share the same underlying likelihood model,[ 18 ] but they are designed to address different types of questions in serology. Thus, these tools provide a complementary suite of methods to tackle different dimensions of serological inference challenges. Previous studies have used Reversible-Jump MCMC methods to infer individual-level infection times and statuses from serological data.[ 15 , 16 ] These approaches model functional forms of antibody kinetics to estimate infection times and/or the force of infection, and subsequently establish relationships between estimated antibody titres at the time of infection and protection against disease. While these studies share similarities, such as the ability to infer multiple infections for an individual, their models and codebases are often tailored specifically to the datasets or research questions they address. As a result, they typically lack publicly available, user-friendly interfaces for broader applications to other datasets. In contrast, the methods implemented in serojump focus on a streamlined design that simplifies inference by only allowing a single infection to be inferred. This simplification enables a flexible and modular approach, allowing users to adapt and extend the model framework with user-written code. Unlike many existing methods, serojump provides a plug-and-play interface, offering general usability without requiring the re-coding of bespoke model structures. This flexibility represents a significant advance in accessibility and usability, facilitating broader adoption and application across different datasets. Our study has limitations. The accuracy of the serojump framework depends on the quality of input data, including the timing, frequency, and noise levels of serological samples. However, simulation studies like the ones presented above can be used to assess the likely performance of the method with a given data structure. Our framework does not currently account for individual-level variation in kinetics (e.g., random effects) or the influence of covariates on antibody kinetics, which would be areas for future development to support research investigating these more detailed features. Scalability and runtime also pose challenges; while effective for smaller datasets, further optimization would be beneficial for applications exceeding 1,000 individuals. Potential extensions to the framework include inferring multiple infections per individual and incorporating multiple biomarkers or antigenically varied pathogens. These additions could enable the exploration of longer-term immunological phenomena (e.g., antigenic seniority)[ 29 ] and improve inference for pathogens like influenza and SARS-CoV-2. However, such extensions would significantly increase the parameter space and computational demands. Strategies such as optimizing samplers or integrating population-level MCMC algorithms (e.g., parallel tempering) could mitigate these challenges and allow for more complex hierarchical frameworks to be evaluated. The development and implementation of serojump address key limitations of previous methods used to infer infection times and statuses from serological data. While Reversible-Jump MCMC and related approaches have long been established for modeling latent infection states, their application has often been constrained by bespoke and complex implementations that lack generalizability. By contrast, serojump introduces a user-friendly, modular framework that simplifies inference to a single infection event while providing the flexibility for users to customize and extend the model for diverse datasets and research questions. This approach not only democratises the use of advanced serological modeling techniques but also lays the foundation for broader applications and collaborative development in the field. The ability to balance simplicity with flexibility ensures that serojump is well-positioned to support future advancements in understanding infection dynamics and serological data interpretation. Therefore, we developed a general-purpose statistical framework that leverages reversible jump MCMC to systematically infer antibody kinetics, infections and their timing using individual-level antibody kinetic for an epidemic outbreak. This tool provides a robust, flexible approach to better inform epidemiological parameters from serological data. By making these advanced methods more accessible and adaptable in the form of an R package serojump , we hope this framework will substantially enhance our ability to track and control infectious diseases in diverse settings. Financial Disclose Statement The study was funded by a United Kingdom Research and Innovation Grant (No. MC_PC_19084). JAH is supported by a Wellcome Trust Early Career Award (grant 225001/Z/22/Z). AJK is supported by National Institutes of Health (1R01AI141534-01A1) and Wellcome (226142/Z/22/Z). Competing Interests The authors declare no competing interests. Data Availability All data produced are available online at https://seroanalytics.org/serojump/ SUPPLEMENTARY FIGURE CAPTIONS Supplementary Methods. Supporting documentation for them methods section Figure S1: Convergence diagnostics for simulated data with COP and 0.1 uncertainty in observational error. (A) Trace plots for fitted parameters (sigma, wane, a, b, c) across four Markov chains, illustrating the mixing and convergence of the parameters. (B) Trace plots for the log posterior across the four chains, showing the variability and stabilization of the log posterior over iterations. (C) Convergence diagnostics for fitted parameters, including effective sample size (ess_bulk, ess_tail) and R hat , which assess the adequacy of sampling and convergence for each parameter. (D) Trace plots for transdimensional convergence of the model dimension, with histogram counts of model dimensions sampled across the chains. (E) Trace plots for transdimensional convergence for the SMI (Structural Model Index) and histogram counts of the log-transformed SMI values across chains. (F) Convergence diagnostics for transdimensional parameters, including effective sample size (ess_bulk, ess_tail) and R hat , summarizing the adequacy of sampling and convergence for the transdimensional space. Figure S2: Convergence diagnostics of infection timing for simulated data with COP and 0.1 uncertainty in observational error. (A) Trace plots for the timing of infection for individuals with posterior P(Z) > 0.5 display estimates across four Markov chains. Each point and its uncertainty interval reflect the sampled infection timing for each individual over iterations. (B) Convergence diagnostics for the timing of infection for individuals with posterior P(Z)>0.5, showing R hat values for each individual. The red dashed line indicates the threshold for R hat =1.1, which marks convergence. Figure S3: Convergence diagnostics for simulated data No COP and 0.1 uncertainty in observational error. (A) Trace plots for fitted parameters (sigma, wane, a, b, c) across four Markov chains, illustrating the mixing and convergence of the parameters. (B) Trace plots for the log posterior across the four chains, showing the variability and stabilization of the log posterior over iterations. (C) Convergence diagnostics for fitted parameters, including effective sample size (ess_bulk, ess_tail) and R hat , which assess the adequacy of sampling and convergence for each parameter. (D) Trace plots for transdimensional convergence of the model dimension, with histogram counts of model dimensions sampled across the chains. (E) Trace plots for transdimensional convergence for the SMI (Structural Model Index) and histogram counts of the log-transformed SMI values across chains. (F) Convergence diagnostics for transdimensional parameters, including effective sample size (ess_bulk, ess_tail) and R hat , summarizing the adequacy of sampling and convergence for the transdimensional space. Figure S4: Convergence diagnostics of infection timing for simulated data No COP and 0.1 uncertainty in observational error. (A) Trace plots for the timing of infection for individuals with posterior P(Z) > 0.5 display estimates across four Markov chains. Each point and its uncertainty interval reflect the sampled infection timing for each individual over iterations. (B) Convergence diagnostics for the timing of infection for individuals with posterior P(Z)>0.5, showing R hat values for each individual. The red dashed line indicates the threshold for R hat =1.1, which marks convergence. Figure S5: Convergence diagnostics for empirical data without PCR information (A) Trace plots for fitted parameters (sigma, wane, a, b, c) across four Markov chains, illustrating the mixing and convergence of the parameters. (B) Trace plots for the log posterior across the four chains, showing the variability and stabilization of the log posterior over iterations. (C) Convergence diagnostics for fitted parameters, including effective sample size (ess_bulk, ess_tail) and R hat , which assess the adequacy of sampling and convergence for each parameter. (D) Trace plots for transdimensional convergence of the model dimension, with histogram counts of model dimensions sampled across the chains. (E) Trace plots for transdimensional convergence for the SMI (Structural Model Index) and histogram counts of the log-transformed SMI values across chains. (F) Convergence diagnostics for transdimensional parameters, including effective sample size (ess_bulk, ess_tail) and R hat , summarizing the adequacy of sampling and convergence for the transdimensional space. Figure S6: Convergence diagnostics for empirical data without PCR information. (A) Trace plots for the timing of infection for individuals with posterior P(Z) > 0.5 display estimates across four Markov chains. Each point and its uncertainty interval reflect the sampled infection timing for each individual over iterations. (B) Convergence diagnostics for the timing of infection for individuals with posterior P(Z)>0.5, showing R hat values for each individual. The red dashed line indicates the threshold for R hat =1.1, which marks convergence. Figure S7: Convergence diagnostics for empirical data with PCR information (A) Trace plots for fitted parameters (sigma, wane, a, b, c) across four Markov chains, illustrating the mixing and convergence of the parameters. (B) Trace plots for the log posterior across the four chains, showing the variability and stabilization of the log posterior over iterations. (C) Convergence diagnostics for fitted parameters, including effective sample size (ess_bulk, ess_tail) and R hat , which assess the adequacy of sampling and convergence for each parameter. (D) Trace plots for transdimensional convergence of the model dimension, with histogram counts of model dimensions sampled across the chains. (E) Trace plots for transdimensional convergence for the SMI (Structural Model Index) and histogram counts of the log-transformed SMI values across chains. (F) Convergence diagnostics for transdimensional parameters, including effective sample size (ess_bulk, ess_tail) and R hat , summarizing the adequacy of sampling and convergence for the transdimensional space. Figure S8 Convergence diagnostics for empirical data with PCR. (A) Trace plots for the timing of infection for individuals with posterior P(Z) > 0.5 display estimates across four Markov chains. Each point and its uncertainty interval reflect the sampled infection timing for each individual over iterations. (B) Convergence diagnostics for the timing of infection for individuals with posterior P(Z)>0.5, showing R hat values for each individual. The red dashed line indicates the threshold for R hat =1.1, which marks convergence. Figure S9: Antibody kinetics, recovery of infection timings, and correlates of protection from empirical data without PCR information. (A) Fitted antibody kinetic trajectories for individuals with infection prior to the Delta variant (left), infection during the Delta variant wave (center), and following vaccination (right), showing fold-rise in antibody titre over time for NCP (blue) and spike (orange) biomarkers. Shaded regions represent 95% credible intervals. (B) Inferred epidemic wave during the Delta wave, illustrating the temporal distribution of PCR-confirmed cases (black bars) and total inferred infections (pink shaded curve), with shaded uncertainty. (C) The estimated attack rate during the Delta wave, partitioned into PCR-confirmed cases (dark grey) and total inferred infections (pink). Error bars indicate 95% credible intervals. (D) The fitted logistic curve to the infection risk, showing the posterior probability of infection as a function of antibody titre at infection for NCP (blue) and spike (orange) biomarkers, with shaded regions representing 95% credible intervals. (E) The absolute COP from the fitted infection risk curve and (F) shows the relative COP for the same biomarkers. REFERENCES 1. ↵ Cutts FT , Hanson M . Seroepidemiology: an underused tool for designing and monitoring vaccination programmes in low- and middle-income countries . Trop Med Int Health . 2016 ; 21 : 1086 – 1098 . OpenUrl CrossRef PubMed 2. ↵ Hay J , Routledge I , Takahashi S . Serodynamics: a review of methods for epidemiological inference using serological data . 2023 . doi: 10.31219/osf.io/kqdsn OpenUrl CrossRef 3. ↵ Haselbeck AH , Im J , Prifti K , Marks F , Holm M , Zellweger RM . Serology as a Tool to Assess Infectious Disease Landscapes and Guide Public Health Policy . Pathogens . 2022 ; 11 . doi: 10.3390/pathogens11070732 OpenUrl CrossRef PubMed 4. ↵ Hellewell J , Russell TW , SAFER Investigators and Field Study Team , Crick COVID-19 Consortium, CMMID COVID-19 working group, Beale R, et al. Estimating the effectiveness of routine asymptomatic PCR testing at different frequencies for the detection of SARS-CoV-2 infections . BMC Med. 2021 ; 19 : 106 . OpenUrl CrossRef PubMed 5. ↵ Crawford DH , Macsween KF , Higgins CD , Thomas R , McAulay K , Williams H , et al. A cohort study among university students: identification of risk factors for Epstein-Barr virus seroconversion and infectious mononucleosis . Clin Infect Dis. 2006 ; 43 : 276 – 282 . OpenUrl CrossRef PubMed Web of Science 6. Wansom T , Muangnoicharoen S , Nitayaphan S , Kitsiripornchai S , Crowell TA , Francisco L , et al. Risk Factors for HIV sero-conversion in a high incidence cohort of men who have sex with men and transgender women in Bangkok , Thailand. EClinicalMedicine . 2021 ; 38 : 101033 . 7. ↵ Dhar-Chowdhury P , Paul KK , Haque CE , Hossain S , Lindsay LR , Dibernardo A , et al. Dengue seroprevalence, seroconversion and risk factors in Dhaka, Bangladesh . PLoS Negl Trop Dis . 2017 ; 11 : e0005475 . OpenUrl PubMed 8. ↵ Chan Y , Fornace K , Wu L , Arnold BF , Priest JW , Martin DL , et al. Determining seropositivity-A review of approaches to define population seroprevalence when using multiplex bead assays to assess burden of tropical diseases . PLoS Negl Trop Dis . 2021 ; 15 : e0009457 . OpenUrl CrossRef PubMed 9. van den Berg OE , Stanoeva KR , Zonneveld R , Hoek-van Deursen D , van der Klis FR , van de Kassteele J , et al. Seroprevalence of Toxoplasma gondii and associated risk factors for infection in the Netherlands: third cross-sectional national study . Epidemiol Infect . 2023 ; 151 : e136 . OpenUrl 10. ↵ Colton H , Hodgson D , Hornsby H , Brown R , Mckenzie J , Bradley KL , et al. Risk factors for SARS-CoV-2 seroprevalence following the first pandemic wave in UK healthcare workers in a large NHS Foundation Trust . Wellcome Open Res . 2021 ; 6 : 220 . 11. ↵ Xu C , Liu L , Ren B , Dong L , Zou S , Huang W , et al. Incidence of influenza virus infections confirmed by serology in children and adult in a suburb community, northern China, 2018-2019 influenza season . Influenza Other Respi Viruses . 2021 ; 15 : 262 – 269 . OpenUrl 12. ↵ Cauchemez S , Horby P , Fox A , Mai LQ , Thanh LT , Thai PQ , et al. Influenza infection rates, measurement errors and the interpretation of paired serology . PLoS Pathog . 2012 ; 8 : e1003061 . OpenUrl CrossRef PubMed 13. ↵ Borremans B , Hens N , Beutels P , Leirs H , Reijniers J . Estimating time of infection using prior serological and individual information can greatly improve incidence estimation of human and wildlife infections . PLoS Comput Biol . 2016 ; 12 : e1004882 . OpenUrl CrossRef PubMed 14. ↵ Pepin KM , Kay SL , Golas BD , Shriner SS , Gilbert AT , Miller RS , et al. Inferring infection hazard in wildlife populations by linking data across individual and population scales . Ecol Lett . 2017 ; 20 : 275 – 292 . OpenUrl CrossRef PubMed 15. ↵ Salje H , Cummings DAT , Rodriguez-Barraquer I , Katzelnick LC , Lessler J , Klungthong C , et al. Reconstruction of antibody dynamics and infection histories to evaluate dengue risk . Nature . 2018 ; 557 : 719 – 723 . OpenUrl CrossRef PubMed 16. ↵ Tsang TK , Perera RAPM , Fang VJ , Wong JY , Shiu EY , So HC , et al. Reconstructing antibody dynamics to estimate the risk of influenza virus infection . Nat Commun . 2022 ; 13 : 1557 . OpenUrl PubMed 17. ↵ Green PJ . Reversible jump Markov chain Monte Carlo computation and Bayesian model determination . Biometrika . 1995 ; 82 : 711 – 732 . OpenUrl CrossRef Web of Science 18. ↵ Menezes A , Takahashi S , Routledge I , Metcalf CJE , Graham AL , Hay JA. serosim: An R package for simulating serological data arising from vaccination, epidemiological and antibody kinetics processes . PLoS Comput Biol . 2023 ; 19 : e1011384 . OpenUrl PubMed 19. ↵ Hay JA , Minter A , Ainslie KEC , Lessler J , Yang B , Cummings DAT , et al. An open source tool to infer epidemiological and immunological dynamics from serological data: serosolver . PLoS Comput Biol . 2020 ; 16 : e1007840 . OpenUrl CrossRef PubMed 20. ↵ Hobson D , Curry RL , Beare AS , Ward-Gardner A . The role of serum haemagglutination-inhibiting antibody in protection against challenge infection with influenza A2 and B viruses . J Hyg . 1972 ; 70 : 767 – 777 . OpenUrl CrossRef PubMed 21. ↵ Khoury DS , Cromer D , Reynaldi A , Schlub TE , Wheatley AK , Juno JA , et al. Neutralizing antibody levels are highly predictive of immune protection from symptomatic SARS-CoV-2 infection . Nat Med . 2021 ; 27 : 1205 – 1211 . OpenUrl CrossRef PubMed 22. ↵ Srivastava K , Carreño JM , Gleason C , Monahan B , Singh G , Abbad A , et al. SARS-CoV-2-infection- and vaccine-induced antibody responses are long lasting with an initial waning phase followed by a stabilization phase . Immunity . 2024 ; 57 : 587 – 599 .e4. OpenUrl CrossRef PubMed 23. ↵ Jarju S , Wenlock RD , Danso M , Jobe D , Jagne YJ , Darboe A , et al. High SARS-CoV-2 incidence and asymptomatic fraction during Delta and Omicron BA.1 waves in The Gambia . Nat Commun . 2024 ; 15 : 3814 . OpenUrl PubMed 24. ↵ Simonsen J , Mølbak K , Falkenhorst G , Krogfelt KA , Linneberg A , Teunis PFM . Estimation of incidences of infectious diseases based on antibody measurements . Stat Med . 2009 ; 28 : 1882 – 1895 . OpenUrl CrossRef PubMed 25. ↵ R Core Team . R: A Language and Environment for Statistical Computing. Vienna, Austria: R Foundation for Statistical Computing ; 2016 . Available: http://www.r-project.org/ 26. ↵ Seamless R and C++ Integration [R package Rcpp version 1.0.14] . 2025 [cited 4 Mar 2025]. Available: https://CRAN.R-project.org/package=Rcpp 27. ↵ Bayart J-L , Morimont L , Closset M , Wieërs G , Roy T , Gerin V , et al. Confounding factors influencing the kinetics and magnitude of serological response following administration of BNT162b2 . Microorganisms . 2021 ; 9 : 1340 . OpenUrl PubMed 28. ↵ Błaszczuk A , Michalski A , Malm M , Drop B , Polz-Dacewicz M . Antibodies to NCP, RBD and S2 SARS-CoV-2 in vaccinated and unvaccinated healthcare workers . Vaccines (Basel ). 2022 ; 10 : 1169 . OpenUrl PubMed 29. ↵ Lessler J , Riley S , Read JM , Wang S , Zhu H , Smith GJD , et al. Evidence for antigenic seniority in influenza A (H3N2) antibody responses in southern China . PLoS Pathog . 2012 ; 8 : e1002802 . OpenUrl CrossRef PubMed View the discussion thread. Back to top Previous Next Posted March 05, 2025. Download PDF Supplementary Material Data/Code Email Thank you for your interest in spreading the word about medRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following serojump: A Bayesian tool for inferring infection timing and antibody kinetics from longitudinal serological data Message Subject (Your Name) has forwarded a page to you from medRxiv Message Body (Your Name) thought you would like to see this page from the medRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share serojump : A Bayesian tool for inferring infection timing and antibody kinetics from longitudinal serological data David Hodgson , James Hay , Sheikh Jarju , Dawda Jobe , Rhys Wenlock , Thushan I. de Silva , Adam J. Kucharski medRxiv 2025.03.04.25323335; doi: https://doi.org/10.1101/2025.03.04.25323335 Share This Article: Copy Citation Tools serojump : A Bayesian tool for inferring infection timing and antibody kinetics from longitudinal serological data David Hodgson , James Hay , Sheikh Jarju , Dawda Jobe , Rhys Wenlock , Thushan I. de Silva , Adam J. Kucharski medRxiv 2025.03.04.25323335; doi: https://doi.org/10.1101/2025.03.04.25323335 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Infectious Diseases (except HIV/AIDS) Subject Areas All Articles Addiction Medicine (568) Allergy and Immunology (863) Anesthesia (300) Cardiovascular Medicine (4436) Dentistry and Oral Medicine (444) Dermatology (382) Emergency Medicine (608) Endocrinology (including Diabetes Mellitus and Metabolic Disease) (1509) Epidemiology (15229) Forensic Medicine (30) Gastroenterology (1124) Genetic and Genomic Medicine (6600) Geriatric Medicine (668) Health Economics (997) Health Informatics (4538) Health Policy (1368) Health Systems and Quality Improvement (1613) Hematology (542) HIV/AIDS (1264) Infectious Diseases (except HIV/AIDS) (15916) Intensive Care and Critical Care Medicine (1103) Medical Education (623) Medical Ethics (146) Nephrology (667) Neurology (6599) Nursing (346) Nutrition (998) Obstetrics and Gynecology (1144) Occupational and Environmental Health (957) Oncology (3333) Ophthalmology (974) Orthopedics (369) Otolaryngology (420) Pain Medicine (436) Palliative Medicine (130) Pathology (663) Pediatrics (1693) Pharmacology and Therapeutics (691) Primary Care Research (711) Psychiatry and Clinical Psychology (5447) Public and Global Health (9232) Radiology and Imaging (2198) Rehabilitation Medicine and Physical Therapy (1370) Respiratory Medicine (1196) Rheumatology (593) Sexual and Reproductive Health (712) Sports Medicine (530) Surgery (712) Toxicology (99) Transplantation (289) Urology (265) (function(){function c(){var b=a.contentDocument||a.contentWindow.document;if(b){var d=b.createElement('script');d.innerHTML="window.__CF$cv$params={r:'a00f4af79c9a4ed4',t:'MTc3OTY1NzA4Ng=='};var a=document.createElement('script');a.src='/cdn-cgi/challenge-platform/scripts/jsd/main.js';document.getElementsByTagName('head')[0].appendChild(a);";b.getElementsByTagName('head')[0].appendChild(d)}}if(document.body){var a=document.createElement('iframe');a.height=1;a.width=1;a.style.position='absolute';a.style.top=0;a.style.left=0;a.style.border='none';a.style.visibility='hidden';document.body.appendChild(a);if('loading'!==document.readyState)c();else if(window.addEventListener)document.addEventListener('DOMContentLoaded',c);else{var e=document.onreadystatechange||function(){};document.onreadystatechange=function(b){e(b);'loading'!==document.readyState&&(document.onreadystatechange=e,c())}}}})();
Text is read by the "Ask this paper" AI Q&A widget below.
Extraction quality varies by source — PMC NXML preserves structure
cleanly, OA-HTML may include some navigation residue, and OA-PDF can
have broken hyphenation. The publisher copy
(via DOI)
is the canonical version.