A multi-region discrete time chain binomial model for infectious disease transmission

preprint OA: closed
📄 Open PDF Full text JSON View at publisher

Abstract

Conventional mathematical models of infectious diseases often overlook the spatial spread of the disease, focusing only on local transmission. However, spatial propagation of various diseases has been observed mainly due to the movement of infectious individuals from one geographical region to another. In this work, we propose a multiregion discrete-time chain binomial framework to model the dependencies between the multiple infection time series from neighboring regions. It is assumed that infection counts in each region at various time points are governed not only by local transmissions but also by interactions of individuals between spatial units. The effects of intervention strategies such as vaccination campaigns used in disease control and various other sociodemographic factors such as live births, population density, vaccination coverage, and disruption in disease surveillance have been taken into account while modeling multiple infection time series. For estimating these multiregion chain-binomial models, an appropriate likelihood function maximization approach is proposed. Simulation results considering effects of seasonal pattern in disease outbreaks, out-of-sync outbreaks in connected geographical regions, variations in interventions have been considered to depict realistic disease scenarios. Forecasting of spatial infections have also been studied. A real world application based on measles counts from adjoining spatial regions is presented to motivate the proposed modeling approach.
Full text 53,662 characters · extracted from preprint-html · click to expand
A multi-region discrete time chain binomial model for infectious disease transmission | medRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-P4HH5NV'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search A multi-region discrete time chain binomial model for infectious disease transmission Pallab K. Sinha , Siuli Mukhopadhyay doi: https://doi.org/10.1101/2025.05.26.25328280 Pallab K. Sinha Department of Mathematics, Indian Institute of Technology Bombay , Maharashtra, India Find this author on Google Scholar Find this author on PubMed Search for this author on this site Siuli Mukhopadhyay Department of Mathematics, Indian Institute of Technology Bombay , Maharashtra, India Find this author on Google Scholar Find this author on PubMed Search for this author on this site For correspondence: siuli{at}math.iitb.ac.in Abstract Full Text Info/History Metrics Data/Code Preview PDF Abstract Conventional mathematical models of infectious diseases often overlook the spatial spread of the disease, focusing only on local transmission. However, spatial propagation of various diseases has been observed mainly due to the movement of infectious individuals from one geographical region to another. In this work, we propose a multiregion discrete-time chain binomial framework to model the dependencies between the multiple infection time series from neighboring regions. It is assumed that infection counts in each region at various time points are governed not only by local transmissions but also by interactions of individuals between spatial units. The effects of intervention strategies such as vaccination campaigns used in disease control and sociodemographic factors such as live births and population density were taken into account while modeling multiple infection time series. For estimating these multiregion chain-binomial models, an appropriate likelihood function maximization approach is proposed. A predictive likelihood based method is used to generate short- and long term forecasts of disease counts in connected spatial units. In-depth analyses of two real world datasets and simulation studies—incorporating seasonal patterns, asynchronous outbreaks across connected regions, and varying intervention strategies—are used to motivate the proposed modeling framework and demonstrate its ability to reproduce realistic outbreak dynamics. 1 Introduction Spatio-temporal models of infectious diseases are essential to improve our understanding of epidemic dynamics as they are able to capture the simultaneous evolution of the disease over time and space. The spatial propagation of various infectious diseases has been documented multiple times throughout history in the epidemiological literature, for example, the 14th century plague in Europe, smallpox in the 1500s, measles in the United Kingdom in 1900s ( Grenfell et al. (2001) ), and most recently the Covid-19 pandemic ( Kianfar et al. (2022) ). This spatial propagation of infectious diseases can be attributed to the intermixture of infectious individuals from one region with healthy susceptibles from another region. In this work, our aim is to propose a novel spatio-temporal epidemiological model that can accurately predict and forecast counts of an infectious disease, in a multi-region setup. We integrate traditional epidemiological model, the chain binomial framework, with statistical techniques to develop a new approach that offers a comprehensive view of the disease dynamics over space and time. Epidemic chain models of infectious disease transmission, such as the discrete-time chain binomial model, have been widely used in epidemiological studies to understand transmission of various diseases such as measles in a school setting ( En’Ko (1989) ), household chickenpox transmission ( Abbey (1952) ), and ebola, swine fever and foot and mouth diseases ( Ferrari et al., 2005 ). In these models ( Bailey (1957) , Becker (2017) ), the binomial formulation arises from the assumption that each potential exposure during an epidemic is an independent Bernoulli trial, resulting in either infection or continued susceptibility. The Reed–Frost model, a well-known example of a chain binomial model, assumes that the number of new infections in time t depends on the number of infectious and susceptible individuals at the previous time point t − 1. The chain structure of these models makes them suitable for parameter estimation using maximum likelihood methods. Recently, chain binomial models have been utilized in the epidemic modeling of COVID-19, where (Lefévre et al., 2021) formulated a generalized Reed–Frost model that included two classes of infections (symptomatic vs asymptomatic) for testing and vaccination, However, prior to this work, epidemic chain models have not been utilized to examine the spatial dynamics of infectious disease transmission. To model the spatial transmission of infections caused by movement of infectious individuals from one region coming into effective contact with the susceptibles of another region, we propose a new multi-region chain binomial framework. This model is based on the epidemiological principle that new infections are generated from the corresponding susceptible populations and previous infection counts. To enhance the statistical rigor of the proposed approach, we assume that the probability of new infections for each spatial unit are modeled through an appropriately chosen link function to depend on the lagged local and inter-regional incidence and region-specific time-dependent covariate effects. Instead of a simple auto regressive order 1 (AR(1)) structure as in conventional chain models, we allow for AR(p) dependence. To address the rapid growth in model complexity as the number of spatial units increases, we incorporate spatial contiguity matrices and k-nearest-neighbor structures to impose sparsity on inter-regional transmission effects. In-depth simulations are conducted to study the effect of spatial distances, inter-regional infection counts, disease transmission rates, and vaccine coverages on the performance of the proposed method under various scenarios depicting real life disease epidemics. We also illustrate the performance of the proposed model using two real world data sets. The first dataset consists of city-level measles incidence data from the United Kingdom spanning the period 1944–1966. This dataset represents the pre-vaccination era and has been extensively analyzed in the epidemiological and statistical literature. Seasonal variation, the baby boom effect, and city-specific live birth rates are known to be important covariates influencing measles dynamics in this period. Accordingly, we assume 2% to 20% of the city-specific populations to represent the corresponding susceptible populations. To demonstrate the applicability of the proposed method in the post-vaccination era, we further analyze a more recent district-level measles dataset from an Indian state covering the period 2014–2020. Due to the effect of vaccination campaigns, the second data set allows us to adjust the susceptible population of each district by accounting for vaccinated cohorts over time. For each real data set, comparison and subsequent selection of models are carried out by using various criterion functions such as mean squared errors based on various error residuals, chi-squared statistic, deviance, and Akaike Information criterion (AIC). We also use the predictive likelihood method ( Mukhopadhyay and Sathish (2019) ) to simultaneously forecast disease counts in all spatial units involved for both data sets. Both short-term and long-term forecasting performance are examined, and the corresponding summary metrics are shown to be related to population density and public transport efficiency within the spatial units considered. Although both of our real-world datasets are based on measles epidemics, the proposed approach is applicable to a broad class of directly transmitted diseases in which infection results in either long-term immunity or death, thereby requiring consideration of only susceptible and infectious compartments. Furthermore, diseases that progress rapidly without long latent periods, and therefore do not need explicit modeling of host birth and death dynamics in the equations ( Ferrari et al., 2005 ), can also be effectively modeled. Various spatio-temporal models have been proposed in the literature to study the dynamics of disease epidemics. Building on the SIR model of ( Kermack and McKendrick, 1927 ), Hethcote (2000) , Satten-spiel and Dietz (1995), Zakary et al. (2017) proposed a multiregion version of the discrete SIR model that accounts for the spread of infectious diseases across multiple geographical regions, incorporating the movement of the population between them. A time-series adaptation of the SIR model, known as the TSIR model ( Grenfell et al., 2001 ), has been used extensively to model the spatial spread of diseases. Bjørnstad et al. (2002) examined measles transmission in connected regions, modeling the force of infection as a function of resident infectious hosts, immigrant infections, and the transmission rate, while, Xia et al. (2004) integrated the TSIR model with a gravity-based spatial interaction framework to describe measles transmission between different communities. Another series of models, namely the endemic-epidemic models ( Held et al. (2005) ) or the HHH models have been used to study infectious diseases over space and time. These models separate infection dynamics into two components: an endemic component handling the influx of new infections from external sources, and an epidemic component covering the contagious nature of the disease by assuming the expected number of transmissions to be a function of earlier infections in time. Meyer and Held (2017) studied this framework in a spatio-temporal setting, modeling case counts at specific locations and times as Poisson or negative binomial variables, with means determined by autoregressive effects, spatial transmission, and time- or unit-specific factors. Similarly, Bauer and Wakefield (2018) modeled counts using a negative binomial distribution and incorporated random endemic, neighborhood, and spatial CAR effects. More recently, Bracher and Held (2022) explored alternative weighting schemes—such as shifted Poisson, linear decay, and geometric weights—while Lu and Meyer (2023) introduced a zero-inflated version of the HHH model to address excess zeros. It should be noted that the spatial versions of the TSIR models are not flexible enough to model AR(p) dependence on local or inter region counts or include the effect of region specific covariates which may influence the dynamics of the disease process. HHH models on the other hand, consider covariate effects but do not incorporate the epidemiological information of the disease, i.e., the generation of new cases through effective contact between infectious and susceptible individuals. In contrast, our proposed multi-region chain binomial model proposes an efficient way to combine epidemiological aspects of a disease with statistical modeling. It allows for the incorporation of various covariate factors that influence disease outbreaks in a statistically rigorous manner. In addition, the chain binomial model provides a more granular understanding of individual infection events, enabling precise estimation of transmission probabilities and the identification of key drivers of outbreaks. This level of detail is especially important for designing targeted interventions and understanding the early spread of infectious diseases. The remainder of the manuscript is structured as follows. Section 2 introduces the chain binomial model. In Section 3 we provide details on the proposed multivariate chain binomial model, along with parameter estimation, model selection statistics and a forecasting algorithm. In Sections 4 and 5 , we give simulation results and analyse two real-world datasets, respectively, to illustrate the application of the proposed approach to model disease epidemics over space and time. Finally, Section 6 provides closing remarks. 2 The chain binomial model - single region We start by introducing the conventional chain binomial model and show steps to add statistical rigour. Suppose, a time series of disease counts have been obtained from a single region. The number of infections at time ( t + 1) is modelled as: where S t is the number of people at risk or susceptible to the disease at time t . The probability of infection π t +1 is generally chosen as a suitable nondecreasing function of the number of infections at time t , that is, I t , between 0 and 1. The susceptibles at time t + 1 is where λ t is the force of infection and can be written as λ t = µI t , µ is the effective rate of contact among infectious and susceptible individuals of that region. Thus, we obtain a generative equation , where S 0 is the initial size of the susceptible population. We deviate from a Markovian type of dependence as in ( Ferrari et al., 2005 ), and assume that the probability of infections at time t may depend on infections until time lag p as in an AR( p ) process. Also, we use a link function approach to incorporate the effect of long term disease trend, seasonal fluctuations, and time dependent covariates to model the probability of new infection. Consider that the conditional distribution of I t +1 | H t is assumed to follow a Bin where H t = { x t , S 0 , I t , …, I 1 } . A logit link function is used as follows. with , z t = ( x t , I t −1 , …, I t − d ), and g d () is any known function for each d (Fokianos Kedem and (2004)). In the above equation, is assumed to be of the form A t + B t + C t , where A t represents a possible long-term temporal trend component, B t the seasonal components, and C t may involve time dependent covariates such as vaccination coverages over time, live births, etc. The autoregressive part of the linear predictor is represented by the p terms, . Similar forms of η t have been used in the endemic-epidemic models as described in Held et al. (2005) . If required, lag terms corresponding to time dependent covariates may also be introduced. As an example, see lag terms of weather variables in the vector-borne disease model in Mukhopadhyay et al. (2019) . 2.1 A new multi-region chain binomial model To extend the conventional chain binomial model to a multi region model we make use of the multiple SIR type of framework as discussed in Zakary et al. (2017) . Suppose that there are m spatial regions and we observe disease counts , from the j th location, where j = 1, …, m . The susceptible population in the region j can then be written as , where is the region-specific force of infection and µ jj ′ is the effective rate of contact between susceptible individuals in the region j with infectious individuals j ′ . Thus, the multi-region chain binomial model is as follows, where the generative equation for the susceptible population j th is updated to , and now represents the probability of infection at time ( t + 1) in the region j . As the m incidence counts observed are from neighboring spatial regions, they may be correlated with each other. Thus, we need to accommodate these dependencies over both time and space while modelling the probability of new infections in region j . We assume that the probability of new infection in region j , , has AR( p ) dependence not only on the disease counts of the same region j , i.e., , but also on the counts of the neighboring regions of j , that is, , where j ′ ∈ N j , N j is a spatial neighborhood of j . Thus, the conditional distribution of is as follows: where , for all j ′ ∈ N j . Using a logit link function we model as Here, we propose two choices to define N j . Choice 1: N j = j ′ : j ′ ≠ j , that is, assume that all surrounding regions may affect the counts of region j irrespective of sharing boundaries or being close to each other. Choice 2: N j = {j ′ : j ′ chosen using k -nearest neighbour rule } , for pre-specified values of k . For Choice 1, note that even for 3 spatial regions and a AR(2) dependence, for each j we have 6 different s, i.e. total 18 parameters. To reduce the number of inter region parameters, , we assume that ′ , where w jj ′ are distance-based weights. Thus, we introduce a j th region specific effect and weight it by its distance from its neighbour j ′ . We may model w jj ′ as a decreasing function of the distance between the centroids of the spatial regions j and j ′ , one popular choice is , and ρ may be estimated from the data. For estimating the decay parameter, ρ , it is possible to use shifted poisson weights, linear decay or geometric weighting scheme as mentioned in Bracher and Held (2022) . However, for computational simplicity, we have fixed the value of ρ to 2 in this manuscript. Note, d jj = 0. For our computations, we also tried another set of weights w jj ′ = exp(− d jj ′ ). See Section 4 for details. Thus, using the distance-based weights the logit-AR(p) model is as follows for region j 2.2 Parameter Estimation We use the maximum likelihood method for estimating the parameters, ν = ( ν 1 , …, ν m ), Using the logit link function, , where can be written as . The Fisher scoring algorithm is used to find ν iteratively as, where ω denotes the ω th iteration step in the estimation process. The score function is, and the conditional covariance matrix G n ( ν ) is given by: where W ( ν ) is a diagonal matrix with elements and . Fokianos and Kedem (1998) showed that where G is the limiting term of G n and c is the dimension of ν . 2.3 Susceptible population The above MLE procedure depends on the values of the initial susceptible populations, . Determining a suitable initial susceptible population S 0 , that best describes the dynamics of the disease, is an important step to correctly analyze disease epidemics. In the epidemiological literature, it is generally assumed that S 0 is the proportion not immune of the total population, i.e., S 0 ≈ (1 − p ) N where p denotes the immunity coverage and N denotes the population ( Anderson and May (1991) , Keeling and Rohani (2002) ). Bjørnstad (2022) used from 2% to 20% of the total population as a candidate of the initial susceptible population. However, other methods such as estimating S 0 using MLE as in Ferrari et al. (2005) have also been used. Very recently, Madden et al. (2024) used the TSIR model to get estimates of susceptibles in their neural network model setup. In this work, we follow the practices of the standard epidemic modeling literature and take , j = 1, …, m to be varying fractions of the region specific populations based on immunity or vaccination coverage. The results represented in later sections are based on 20% of the population densities, as is usually done for measles ( Bjørnstad (2022) , Ferrari et al. (2008) ). In vaccine preventable diseases (VPDs), particularly measles, the generative equation for is further updated as where represents the non-immune birth cohort at time t and region j due to vaccination. For a vaccine preventable disease, if the number of scheduled doses, the efficacy of each dose, and coverages are known, the computation of can be based on this information. Suppose, we have a VPD with one dose of vaccine. Then the non-immune cohort is Chen et al. (2012) : where is the birth cohort, is the coverage of vaccination, and the efficacy of the vaccine is x . 2.4 Model Evaluation Various forms of the above multiregion chain binomial model were fitted to simulated and real data sets of disease counts. For evaluating these models, we used popular goodness-of-fit statistics from the model selection literature like model deviance, chi-squared criterion, AIC, Bayesian Schwarz information criterion (BIC), various residuals (such as working, Pearson etc.) and mean square errors based on these residuals ( Kedem and Fokianos (2005) ). 2.5 Forecasting using a Predictive Likelihood Once the model is selected, it is of interest to forecast future disease incidences and obtain intervals for these forecasts in the m regions. Based on past information up to the time point n , , for all j regions, we estimate the density of the predicted value (one step ahead) using the predictive likelihood method ( Mukhopadhyay and Sathish (2019) ) as where k (·) is a normalizing constant and . The forecast value of denoted by , is arg max , where . Note, involves the counts of the j th region as well as j ′ ∈ N j . To generate, a set of q one-step ahead forecasts, the following steps are used: Step 1: Generate the first forecast for all j regions; For ω = 1 and j = 1, …, m using the predictive likelihood, Step 2: To generate , for ω = 2, …, q and j = 1, …, m we use the predictive likelihood where, for all j ′ ∈ N j . To estimate the prediction interval for a set of one-step forward forecasts, we use the region of highest density of Hyndman (1995) . The (1 − α )100% HDR for , ω = 1, …, q is the region , where , where f α is the largest constant such that is at least (1 − α ). 3 Simulation Studies We perform simulations in two different scenarios to show the applicability of our proposed method to track disease epidemics over space and time when interventions are available for disease control. We assume that the disease being modeled is the biweekly counts of measles. Measles is a highly contagious, vaccine-preventable viral disease. The World Health Organization recommends administering two doses of measles vaccines at 9-12 months and 16-24 months to control and eliminate the disease. Measles have been modeled previously by various researchers, namely, Grenfell et al. (2001) , Chen et al. (2012) , Thakkar et al. (2019) . For generating biweekly counts we use the following susceptible-infection equations of the multiregion SIR model ( Zakary et al. (2017) ) with various choices of parameters. It is assumed that there are 3 spatially connected regions and 160 biweekly incidences are generated using the following equations, where the recovery rate, γ , is assumed to be 1/14 day −1 and µ jk ’s are the disease transmission rates between two locations j and k . The transmission rates are taken to be influenced by region-specific factors such as the number of susceptibles and annual seasonality, here are the region-specific mean transmission rates. We model the µ jk ’s to depend on the distance between two spatial regions j and k as: µ jk ( t ) = µ jj ( t ) w jk , where w jk ’s are spatial weights. A spatial weight matrix, W = (( w jk )), is used. We work with the following parameters choices: Case 1: ; j = 1, 2, 3 and , ∀ j . Case 2: , j = 1, 2; , ∀ j . As interventions such as vaccination strategies play an important role in measles control, we also introduce a vaccinated birth cohort as discussed in Section 2.3 . The susceptible population in the equations of the MSIR model is adjusted as: where The birth cohort was chosen to remain constant over time . Thus, for a region j where vaccination is available, denotes the birth cohort who remains unvaccinated, thus adding to the susceptible population of the region. For each case, we consider two scenarios: In scenario 1 (without vaccination): counts are generated without the effect of vaccination in any of the three spatial regions. In scenario 2 (with vaccination): we assume that vaccination is available in Region 3 only. The coverages of the two doses are, increasing linearly from 0.70 to 0.90 and increasing linearly from 0.40 to 0.70 over the biweeks (t). After fitting the proposed model to the data generated from the MSIR model, the results are plotted in Figure 1 . In both cases we see that counts of regions 2 and 3 behave similarly as these are the spatially coupled regions. Also, for case 2, as the initial case counts of region 3 is higher, it increases the counts of the spatially coupled regions 2 and 3 to be higher than in case 1, while region 1 remains unaffected in both cases. We also note that for both parameter choices (cases 1 and 2), the effect of vaccination reduces the measles counts in the spatially related regions 2 and 3, while region 1 remains unaffected. This reduction may be attributed to the spatial proximity of regions 2 and 3, as vaccination available in region 3 may lead to a lower influx of infected individuals to migrate from Region 3 to Region 2, thus reducing disease transmission. Download figure Open in new tab Figure 1: Comparison of measles incidences in the three regions for the two simulation scenarios (S1-S2) and cases (C1-C2). In each plot, represent simulated counts of region 1, 2, and 3 respectively, and , and represent the estimated counts using the proposed approach in Section 2. For both without and with vaccination setups, 10, 000 simulations were performed and the proposed model in Section 2 was fitted. The mean squared error (MSE) and mean squared absolute deviation (MAE) values are listed in Table 1 . Note that the error statistics values are similar for regions 2 and 3. This is due to the fact that in these scenarios we assume that Regions 2 and 3 are spatially coupled and have comparable incidence patterns. The distribution of the estimated parameters (standardized parameters) and the residual analysis were also studied; however, we do not report them here. View this table: View inline View popup Download powerpoint Table 1: Error Statistics for Different Regions Under Various Simulation Scenarios in Section 3. Scenarios 1 and 2 are indicated as S1 and S2. Cases 1 and 2 are indicated as C1 and C2. 4 Real Data Example: Analysis of Measles Counts in UK To illustrate our proposed model on real data, we fitted it to biweekly case reports of measles recorded between the years 1944-1966 from seven urban cities, Birmingham, Coventry, Leicester, Derby, Nottingham, Wolverhampton and Walsall, in the United Kingdom (UK). The UK measles data set has previously been analyzed by various authors, namely ( Grenfell et al. (2001) , Xia et al. (2004) , Madden et al. (2024) ). We chose the seven cities due to their spatial proximities (distance) on the UK map. From an initial spatial correlation analysis, a Moran’s-I value ( Anselin (1995) ) of 3.0277 and a p-value of 0.0012 (significant at the 5% level) were observed, implying existence of spatial correlation among the disease counts (averaged over time) of the seven cities. During the period 1944 to 1966, these seven English Midlands cities (see Fig. 2 ) were connected through a complex and evolving public transport network Eye (2016) that facilitated the frequent movement of individuals among the cities. The railway system served as the primary inter-city connector, with Birmingham operating as the central hub thus, justifying the highest case counts of Birmingham. For the location of the cities and the strength of the public transport connection between them, see Figure 2 , while for distances between cities, see Table 2 . From Figure 2 , as Birmingham is the most strongly connected to all cities, outbreaks in all cities seem to be synchronized or follow. These seven cities also differ in their population status, with Birmingham having the highest population. We divide the cities that exclude Birmingham into two groups (high/low) according to their population densities and plot the measles counts to compare the incidences and pattern with Birmingham in Fig. 3 . View this table: View inline View popup Download powerpoint Table 2: Distance in km among the seven UK cities for Section 4. Download figure Open in new tab Figure 2: Left panel (a) of the figure shows the location of the seven cities in the UK map and the right panel (b) shows the strength of connection between the seven UK cities. In (b), the red lines represent strong connections, blue lines represent moderate connections and yellow lines represent weak connections. Download figure Open in new tab Figure 3: Outbreak comparisons between Birmingham and other cities on the basis of raw incidences in Section 4 . The cities in (a) are the high population density cities while in (b) we have the low population density cities. The measles data was divided into a training set and a test set. The training sets were chosen as the period 1945 (January) to 1966 (June) and thereafter were the test data. Using various forms of equation 3 , our objective was to select the best model to understand the spatial coupling between these seven UK cities, study the transmission within and between cities through , and the region specific trend, seasonality and covariates which affect the spread of the disease. For this we studied long term trend effect, seasonality (annual/biennial) and covariates such as live births in 1946 and baby boom effect in the model. To select the lag order d , we tested different values of p ranging from 1 to 7, and ranked the models according to their BIC values, the lowest BIC (1286.25) was obtained for p=1. We checked various forms of g (·) in equation 3 ; and, the best fitting results were obtained for the logarithm function. The models chosen based on the training set for each of the 7 cities are described in Table 3 . These models were chosen with respect to various goodness of fitness statistics as mentioned in section 2.4. To see some of the values of the GOF statistics for the chosen models, refer to Figure 4a . View this table: View inline View popup Download powerpoint Table 3: Description of best model for each of the seven UK cities in Section 4. Download figure Open in new tab Figure 4: Prediction and forecasting performance for the seven cities. The left panel (a) shows three GOF statistics, Akaike Information Criterion (AIC), Deviance (D) and Pearson residuals (PR) based on training data. In the right panel (b), we show the forecasting results based on test data. Each dot indicates the RMSE value, and the surrounding whiskers represent the confidence interval of the RMSE. Cities are ordered on the x-axis according to their population. Incidence counts across all spatial units constitute a multivariate time series with inter-dependencies; however, conditional on , the incidence time series for each region can be treated as independent as shown in equation 1 . Consequently, we applied the GLM function from the base stats package in R to each city separately, using the model specification described above. Estimates of model parameters, standard errors, and p-values for each of the seven cities are listed in Table 4 . From Table 4 , we observe an increasing trend in the incidence of measles in all cities. Strong seasonal patterns, both semi-annual and annual, are evident across all locations, indicating periodic oscillations in measles transmission. The covariates related to demographic dynamics, specifically live births and the post-World War-II baby boom, demonstrate significant associations with the incidence of measles. The transmission parameter, , modeling the effect of disease transmission within the city was significant in all cities. We see that the odds of infection increase by a factor of 2.50 for Birmingham, 2.49 for Nottingham, 2.39 for Leicester, 2.43 for Coventry, 2.29 for Wolverhampton, 2.41 for Derby, 2.48 for Walsall. To find , we used for two different weight functions, , and . View this table: View inline View popup Download powerpoint Table 4: Parameter estimates, standard errors and p-values for of the models fitted to the measles counts from all the concerned cities in Section 4. Short- and long term forecasting was performed in the range of 1 month, 6 months, and 12 months for the test data along with 95% intervals using the method described in section 2.5. We note that both in-sample and forecast performance generally deteriorated with city population size ( Figure 4 ). The accuracy of the forecast is evaluated using the root mean square error (RMSE) defined as ( Hyndman and Koehler (2006) ). An HDR confidence interval for RMSE was obtained based on the HDR confidence interval for the predicted disease counts as in section 2.5. 5 Analysis of Measles Counts in West Bengal Since measles is a vaccine-preventable disease, its outbreaks can be largely controlled through immunization. In most parts of the world, measles is now close to eradication due to widespread vaccination and cumulative immunity developed through both vaccination and prior infection. West Bengal, a state in India with an estimated population of 99.36 million, introduced the first dose of the measles-containing vaccine (MCV1) in 1985 and the second dose (MCV2) in 2010 as part of India’s measles elimination program. We extracted monthly aggregated measles cases, and immunization information for first and second doses of measles in all districts of West Bengal from March 2014 to April 2020 from the Health Management Information System (HMIS) database, which is publicly available and downloadable at https://hmis.mohfw.gov.in/#!/standardReports . Although the state currently has 23 districts, it underwent several administrative boundary changes during the study period. The Barddhaman district was divided into Purba Barddhaman and Paschim Barddhaman, and two new districts, Jhargram and Kalimpong, were created. In addition, Alipurduar was separated from Jalpaiguri and established as an independent district in 2015. For district specific measles cases and vaccination coverages, see Figures 5 and 6 . Download figure Open in new tab Figure 5: Normalized measles incidence rates (in monthly scale) for districts in West Bengal (2014-2020) corresponding to Section 5. The grey color indicates either districts that did not exist at the time (e.g., Jhargram, Kalimpong, Alipurduar, Purba Barddhaman and Paschim Barddhaman) or areas with missing data (e.g., Kolkata). Download figure Open in new tab Figure 6: District wise vaccination coverage in West Bengal from 2014 to 2020 corresponding to Section 5. Grey shading indicates periods prior to the initiation of vaccination or missing vaccination data for specific districts. Imputation was used for the missing coverages. As the first dose of the measles vaccine is administered at 9 months of age and the second dose at 16 months of age, to account for the vaccination effect, we again took the unvaccinated cohort at time t in district j as: where denotes the birth cohort in the time period t in district j . The respective vaccine coverages, and are taken from Figure 6 . As discussed in Section 2.3 , the generative equation for the number of susceptibles at time t is updated to . For setting up the spatial neighborhoods, both choices N1 and N2 were used. In the nearest neighbor method (N2) k was taken to be 3. As before, we tested for trend and seasonality terms in our model. We have applied the model described in section 2.1 , with three nearest neighbor districts as a spatial cluster ( N 2 choice). For model training, data from April 2014 to September 2019 were used, while forecasting was done for October 2019 to March 2020. The prediction model for each district was chosen as described based on several GOFs as described in Section 4. The BIC value was reported to be lowest for p = 1 (4329.40). In Figure 7 , we show the medium-term forecast of six months with respect to two types of spatial neighbor choices, N1 and N2. For N 1 , we used w jj ′ = exp(− d jj ′ ). From Figure 7 , we note that the forecast results based on the prediction model with choice N1 outperformed the model that only involved 3 nearest districts ( N 2 ). This is likely to happen due to the well-developed connectivity between the districts of West Bengal. The movement of individuals may not be restricted to the immediate neighboring units and may extend across all districts. Noting that West Bengal’s 3,825-kilometer rail network, mainly run by the Eastern and South Eastern Railway zones, connects its districts efficiently. Important routes like the Howrah–Barddhaman–Asansol and Malda–New Jalpaiguri lines handle key east-west and north-south travel. The widespread Kolkata Suburban Railway serves as a vital daily link for eight surrounding districts. This rail network is complemented by a robust road system, with major National Highways such as NH-34, NH-2, NH-31, and NH-55 ensuring strong connectivity throughout the state. Download figure Open in new tab Figure 7: Comparison of forecasting performances between two choices N1 and N2 for West Bengal district level measles data corresponding to Section 5. Districts are ordered according to their population size. Based on the estimated values of N 1 choice, we sought to identify the dominant inter-district connections influencing measles transmission. Since the choice of N 1 assumes a fully connected network in which all districts are linked, we introduced sparsity by applying a thresholding procedure to the estimated interaction parameters . The estimated values of , for all district pairs j, j ′ , ranged from 0 to 0.237, with a median value of 0.014. We classified a connection as dominant if 0.014; otherwise, the corresponding parameter was set to zero. This thresholding approach enabled the identification of key transmission pathways while reducing weaker connections. The number of dominant edges associated with each district is illustrated in Figure 8 . The districts in the central region of West Bengal exhibit greater connectivity, likely reflecting stronger transportation networks and greater livelihood opportunities that facilitate population movement. In contrast, northern districts show comparatively fewer dominant connections, consistent with migration patterns in which individuals tend to move toward the more economically active central regions of the state. Download figure Open in new tab Figure 8: Connection strength of each district in West Bengal corresponding to Section 5. 6 Concluding Remarks In this study, we developed a novel multivariate model to predict and forecast disease counts based on surveillance data. Our approach integrates advanced time series techniques, specifically the generalized linear auto regressive model, within an epidemiological framework. The model selection process involved evaluating multiple candidate models to identify the one that performed the best. A key limitation of our model is its inability to account for under-reporting, a common issue in surveillance data. Detailed information on under-reporting is generally unavailable. As a future direction, we are currently working to develop a model that uses sero-surveillance data to understand the size of the susceptible population and consequently estimate the underreported cases. While modelling infectious diseases, substantial heterogeneity in incidence is often observed across age groups. This phenomenon is particularly pronounced in childhood infections, where younger children tend to be more vulnerable to infection than older children. In future work, we aim to incorporate such age-specific heterogeneity into our spatio-temporal modelling framework and investigate its impact on parameter estimation as well as short- and long-term forecasting performance. Data Availability All data produced in the present study are available upon request to the authors https://github.com/pallabksinha/Spatio-temporal-modelling Conflict of Interest The authors declare that they have no potential conflict of interest. Disclaimer The work/opinion is based on research findings by the authors and not the opinion of the government. Acknowledgments Funding for this study was provided by the Gates Foundation (INV-044445). We thank Dr. Shubham Niphadkar for his valuable inputs in this project. Footnotes All Sections have been revised. We have added a new section 2.3 to explain the choice of S_0. We have added a new real data section, section 5. We have updated the authors list. The co-author Shubham Niphadkar in the earlier version has been moved to the Acknowledgement section. This has his approval. References ↵ Abbey , H. ( 1952 ). An examination of the Reed-Frost theory of epidemics . Human biology , 24 ( 3 ): 201 . OpenUrl PubMed ↵ Anderson , R. M. and May , R. M. ( 1991 ). Infectious diseases of humans: dynamics and control . Oxford university press . ↵ Anselin , L. ( 1995 ). Local indicators of spatial association—LISA . Geographical Analysis , 27 ( 2 ): 93 – 115 . OpenUrl CrossRef Web of Science ↵ Bailey , N. T. ( 1957 ). The mathematical theory of epidemics . Charles Griffin & Co., Ltd . ↵ Bauer , C. and Wakefield , J. ( 2018 ). Stratified space–time infectious disease modelling, with an application to hand, foot and mouth disease in china . Journal of the Royal Statistical Society Series C: Applied Statistics , 67 ( 5 ): 1379 – 1398 . OpenUrl ↵ Becker , N. G. ( 2017 ). Analysis of infectious disease data . Chapman and Hall/CRC . ↵ Bjørnstad , O. N. ( 2022 ). Epidemics: models and data using R . Springer Nature . ↵ Bjørnstad , O. N. , Finkenstädt , B. F. , and Grenfell , B. T. ( 2002 ). Dynamics of measles epidemics: estimating scaling of transmission rates using a time series SIR model . Ecological Monographs , 72 ( 2 ): 169 – 184 . OpenUrl CrossRef Web of Science ↵ Bracher , J. and Held , L. ( 2022 ). Endemic-epidemic models with discrete-time serial interval distributions for infectious disease prediction . International Journal of Forecasting , 38 ( 3 ): 1221 – 1233 . OpenUrl ↵ Chen , S. , Fricks , J. , and Ferrari , M. J. ( 2012 ). Tracking measles infection through non-linear state space models . Journal of the Royal Statistical Society Series C: Applied Statistics , 61 ( 1 ): 117 – 134 . OpenUrl ↵ En’Ko , P. ( 1989 ). On the course of epidemics of some infectious diseases . ↵ Eye , H. ( 2016 ). The black country’s lost railway network. Accessed : 2025-11-04 . ↵ Ferrari , M. J. , Bjørnstad , O. N. , and Dobson , A. P. ( 2005 ). Estimation and inference of R0 of an infectious pathogen by a removal method . Mathematical Biosciences , 198 ( 1 ): 14 – 26 . OpenUrl CrossRef PubMed Web of Science ↵ Ferrari , M. J. , Grais , R. F. , Bharti , N. , Conlan , A. J. , Bjørnstad , O. N. , Wolfson , L. J. , Guerin , P. J. , Djibo , A. , and Grenfell , B. T. ( 2008 ). The dynamics of measles in sub-saharan africa . Nature , 451 ( 7179 ): 679 – 684 . OpenUrl CrossRef PubMed Web of Science ↵ Fokianos , K. and Kedem , B. ( 1998 ). Prediction and classification of non-stationary categorical time series . Journal of Multivariate Analysis , 67 ( 2 ): 277 – 296 . OpenUrl Fokianos , K. and Kedem , B. ( 2004 ). Partial likelihood inference for time series following generalized linear models . Journal of Time Series Analysis , 25 ( 2 ): 173 – 197 . OpenUrl ↵ Grenfell , B. T. , Bjørnstad , O. N. , and Kappey , J. ( 2001 ). Travelling waves and spatial hierarchies in measles epidemics . Nature , 414 ( 6865 ): 716 – 723 . OpenUrl CrossRef PubMed Web of Science ↵ Held , L. , Höhle , M. , and Hofmann , M. ( 2005 ). A statistical framework for the analysis of multivariate infectious disease surveillance counts . Statistical Modelling , 5 ( 3 ): 187 – 199 . OpenUrl CrossRef Web of Science ↵ Hethcote , H. W. ( 2000 ). The mathematics of infectious diseases . SIAM Review , 42 ( 4 ): 599 – 653 . OpenUrl CrossRef PubMed ↵ Hyndman , R. J. ( 1995 ). Highest-density forecast regions for nonlinear and non-normal time series models . Journal of Forecasting , 14 ( 5 ): 431 – 441 . OpenUrl ↵ Hyndman , R. J. and Koehler , A. B. ( 2006 ). Another look at measures of forecast accuracy . International journal of forecasting , 22 ( 4 ): 679 – 688 . OpenUrl CrossRef Web of Science ↵ Kedem , B. and Fokianos , K. ( 2005 ). Regression models for time series analysis . John Wiley & Sons . ↵ Keeling , M. J. and Rohani , P. ( 2002 ). Estimating spatial coupling in epidemiological systems: a mechanistic approach . Ecology Letters , 5 ( 1 ): 20 – 29 . OpenUrl CrossRef Web of Science ↵ Kermack , W. O. and McKendrick , A. G. ( 1927 ). A contribution to the mathematical theory of epidemics . Proceedings of the Royal Society of London. Series A, Containing papers of a mathematical and physical character , 115 ( 772 ): 700 – 721 . OpenUrl ↵ Kianfar , N. , Mesgari , M. S. , Mollalo , A. , and Kaveh , M. ( 2022 ). Spatio-temporal modeling of covid-19 prevalence and mortality using artificial neural network algorithms . Spatial and Spatio-temporal Epidemiology , 40 : 100471 . OpenUrl Lefèvre , C. , Picard , P. , Simon , M. , and Utev , S. ( 2021 ). A chain binomial epidemic with asymptomatics motivated by covid-19 modelling . Journal of Mathematical Biology , 83 ( 5 ): 54 . OpenUrl PubMed ↵ Lu , J. and Meyer , S. ( 2023 ). A zero-inflated endemic–epidemic model with an application to measles time series in Germany . Biometrical Journal , 65 ( 8 ): 2100408 . OpenUrl ↵ Madden , W. G. , Jin , W. , Lopman , B. , Zufle , A. , Dalziel , B. , E. Metcalf , C. J. , Grenfell , B. T. , and Lau , M. S. ( 2024 ). Deep neural networks for endemic measles dynamics: Comparative analysis and integration with mechanistic models . PLoS computational biology , 20 ( 11 ): e1012616 . OpenUrl ↵ Meyer , S. and Held , L. ( 2017 ). Incorporating social contact data in spatio-temporal models for infectious disease spread . Biostatistics , 18 ( 2 ): 338 – 351 . OpenUrl PubMed ↵ Mukhopadhyay , S. and Sathish , V. ( 2019 ). Predictive likelihood for coherent forecasting of count time series . Journal of Forecasting , 38 ( 3 ): 222 – 235 . OpenUrl ↵ Mukhopadhyay , S. , Tiwari , R. , Shetty , P. , Gogtay , N. , and Thatte , U. ( 2019 ). Modeling and forecasting Indian malaria incidence using generalized time series models . Communications in Statistics: Case Studies, Data Analysis and Applications , 5 ( 2 ): 111 – 120 . OpenUrl Sattenspiel , L. and Dietz , K. ( 1995 ). A structured epidemic model incorporating geographic mobility among regions . Mathematical Biosciences , 128 ( 1-2 ): 71 – 91 . OpenUrl CrossRef PubMed Web of Science ↵ Thakkar , N. , Gilani , S. S. A. , Hasan , Q. , and McCarthy , K. A. ( 2019 ). Decreasing measles burden by optimizing campaign timing . Proceedings of the National Academy of Sciences , 116 ( 22 ): 11069 – 11073 . OpenUrl Abstract / FREE Full Text ↵ Xia , Y. , Bjørnstad , O. N. , and Grenfell , B. T. ( 2004 ). Measles metapopulation dynamics: a gravity model for epidemiological coupling and dynamics . The American Naturalist , 164 ( 2 ): 267 – 281 . OpenUrl CrossRef PubMed Web of Science ↵ Zakary , O. , Rachik , M. , and Elmouki , I. ( 2017 ). On the analysis of a multi-regions discrete SIR epidemic model: an optimal control approach . International Journal of Dynamics and Control , 5 : 917 – 930 . OpenUrl CrossRef View the discussion thread. Back to top Previous Next Posted February 28, 2026. Download PDF Data/Code Email Thank you for your interest in spreading the word about medRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following A multi-region discrete time chain binomial model for infectious disease transmission Message Subject (Your Name) has forwarded a page to you from medRxiv Message Body (Your Name) thought you would like to see this page from the medRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share A multi-region discrete time chain binomial model for infectious disease transmission Pallab K. Sinha , Siuli Mukhopadhyay medRxiv 2025.05.26.25328280; doi: https://doi.org/10.1101/2025.05.26.25328280 Share This Article: Copy Citation Tools A multi-region discrete time chain binomial model for infectious disease transmission Pallab K. Sinha , Siuli Mukhopadhyay medRxiv 2025.05.26.25328280; doi: https://doi.org/10.1101/2025.05.26.25328280 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Infectious Diseases (except HIV/AIDS) Subject Areas All Articles Addiction Medicine (571) Allergy and Immunology (864) Anesthesia (302) Cardiovascular Medicine (4446) Dentistry and Oral Medicine (444) Dermatology (383) Emergency Medicine (609) Endocrinology (including Diabetes Mellitus and Metabolic Disease) (1515) Epidemiology (15240) Forensic Medicine (30) Gastroenterology (1130) Genetic and Genomic Medicine (6612) Geriatric Medicine (669) Health Economics (1000) Health Informatics (4549) Health Policy (1370) Health Systems and Quality Improvement (1613) Hematology (543) HIV/AIDS (1267) Infectious Diseases (except HIV/AIDS) (15926) Intensive Care and Critical Care Medicine (1105) Medical Education (624) Medical Ethics (147) Nephrology (668) Neurology (6619) Nursing (346) Nutrition (999) Obstetrics and Gynecology (1147) Occupational and Environmental Health (957) Oncology (3341) Ophthalmology (976) Orthopedics (369) Otolaryngology (420) Pain Medicine (436) Palliative Medicine (130) Pathology (665) Pediatrics (1695) Pharmacology and Therapeutics (693) Primary Care Research (714) Psychiatry and Clinical Psychology (5458) Public and Global Health (9245) Radiology and Imaging (2205) Rehabilitation Medicine and Physical Therapy (1371) Respiratory Medicine (1197) Rheumatology (597) Sexual and Reproductive Health (715) Sports Medicine (530) Surgery (714) Toxicology (99) Transplantation (289) Urology (265) (function(){function c(){var b=a.contentDocument||a.contentWindow.document;if(b){var d=b.createElement('script');d.innerHTML="window.__CF$cv$params={r:'a0270381af7adf94',t:'MTc3OTkwNTgxNw=='};var a=document.createElement('script');a.src='/cdn-cgi/challenge-platform/scripts/jsd/main.js';document.getElementsByTagName('head')[0].appendChild(a);";b.getElementsByTagName('head')[0].appendChild(d)}}if(document.body){var a=document.createElement('iframe');a.height=1;a.width=1;a.style.position='absolute';a.style.top=0;a.style.left=0;a.style.border='none';a.style.visibility='hidden';document.body.appendChild(a);if('loading'!==document.readyState)c();else if(window.addEventListener)document.addEventListener('DOMContentLoaded',c);else{var e=document.onreadystatechange||function(){};document.onreadystatechange=function(b){e(b);'loading'!==document.readyState&&(document.onreadystatechange=e,c())}}}})();

Text is read by the "Ask this paper" AI Q&A widget below. Extraction quality varies by source — PMC NXML preserves structure cleanly, OA-HTML may include some navigation residue, and OA-PDF can have broken hyphenation. The publisher copy (via DOI) is the canonical version.

My notes (saved in your browser only)

Ask this paper AI returns verbatim quotes from the full text · source: preprint-html

Answers must be backed by verbatim quotes from this paper's full text. Hallucinated quotes are dropped automatically; if no verbatim passage answers the question, we say so. How this works

Citation neighborhood (no data yet)

We don't have any in-corpus citations linked to this paper yet. This is a recent paper (2025) — citers typically take a year or two to land, and the OpenAlex reference graph may still be filling in.

Source provenance

europepmc
last seen: 2026-05-20T01:45:00.602351+00:00