Full text
44,018 characters
· extracted from
preprint-html
· click to expand
Evaluation Methods for T-association of a Surrogate Endpoint | medRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-P4HH5NV'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search Evaluation Methods for T-association of a Surrogate Endpoint Jo-Ying Hung , Chih-Yuan Hsu , Pei-Fang Su , Yu Shyr doi: https://doi.org/10.1101/2025.08.28.25334653 Jo-Ying Hung 1 Department of Statistics, National Cheng Kung University , Tainan, Taiwan Find this author on Google Scholar Find this author on PubMed Search for this author on this site Chih-Yuan Hsu 2 Department of Biostatistics, Vanderbilt University Medical Center , Nashville, Tennessee, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site Pei-Fang Su 1 Department of Statistics, National Cheng Kung University , Tainan, Taiwan Find this author on Google Scholar Find this author on PubMed Search for this author on this site Yu Shyr 2 Department of Biostatistics, Vanderbilt University Medical Center , Nashville, Tennessee, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site For correspondence: yu.shyr{at}vumc.org Abstract Full Text Info/History Metrics Supplementary material Data/Code Preview PDF Abstract A surrogate endpoint is a biomarker that is reasonably likely to predict clinical benefit and is used as a substitute for a direct measure of clinical benefit under the Food and Drug Administration (FDA) Accelerated Approval pathway. According to FDA guidelines, a valid surrogate endpoint must meet two associations: I-association (the association between the surrogate and true endpoints, such as disease response and overall survival) and T-association (the association between treatment effects on both endpoints, such as odds ratio and hazard ratio). I-association is commonly evaluated, but T-association is often overlooked due to the lack of appropriate statistical methods. Failure to satisfy T-association precludes a biomarker from supporting accelerated approval. To address this gap, we propose a new method to rigorously assess T-association in accordance with FDA guidelines. This method assumes that treatment effects on the surrogate and true endpoints follow a bivariate normal distribution, accounting for both within-study and between-study variances. The key evaluation metric is the correlation coefficient, which quantifies the relationship between treatment effects on both endpoints. Model parameters, including this correlation, are estimated using maximum likelihood, restricted maximum likelihood, and a Bayesian approach. We demonstrate the method using both simulated and real-world data. The method will serve as the statistical foundation that aligns with FDA guidelines and supports future accelerated approvals. The R package to implement the proposed method is available at https://github.com/jybelindahung/T-association . 1 Introduction To support effectiveness claims in new drugs applications to the Food and Drug Administration (FDA), appropriate endpoints must be selected for clinical trials. Endpoints used to assess clinical benefit include both clinical endpoints and surrogate endpoints. Clinical endpoints directly reflect how a patient feels, functions, or survives, and are considered the most reliable measure of treatment benefits. In contrast, surrogate endpoints are biomarkers expected to predict clinical benefit and serve as substitutes for clinical endpoints. Because they can be measured earlier or more easily, surrogate endpoints can potentially shorten trial duration and improve compliance, cost-effectiveness, and ethical considerations. The FDA provides two approval pathways for new drugs: traditional approval and accelerated approval. Traditional approval requires direct evidence of clinical benefit, as demonstrated by clinical endpoints or validated surrogate endpoints. Accelerated approval, in contrast, may be granted on the basis of intermediate clinical endpoints or surrogate endpoints that are reasonably likely to predict clinical benefit [ 1 ]. It allows approval of drugs for serious or life-threatening conditions that address unmet medical need. Since the first establishment of accelerated approval in 1992, the accelerated approval pathway has been increasingly utilized. Figure 1 illustrates the growing number of accelerated approval drugs in recent years based on an FDA report [ 2 ]. The trend reflects the expanding role of accelerated approval in drug development. To ensure regulatory rigor under this pathway, the validation of whether a biomarker is a surrogate endpoint is an important issue. Download figure Open in new tab Figure 1. Number of Drugs Granted Accelerated Approval by Year Range (1992–2024). To determine whether a biomarker can be considered a surrogate endpoint, FDA has outlined statistical principles for validation, including individual-level association (I- association) and trial-level association (T-association)[ 3 ]. I-association assesses whether the biomarker correlates with the true clinical endpoint within individuals, while T- association refers to whether the treatment effect on the biomarker is associated with the treatment effect on the true endpoint across trials. For instance, in the aforementioned FDA guideline [ 3 ], it is discussed that although minimal residual disease (a biomarker that measures tumor burden) is widely used to reflect a patient’s response to treatment, validating it as a surrogate endpoint requires evidence of both I-association and T-association. The definitions of I-association and T-association were first proposed by Buyse et al. [ 4 ]. They defined coefficients of determination estimated within trials and across trials, and proposed that a biomarker can be considered a valid surrogate only if it demonstrates both individual- and trial-level validity. Despite the rigorous definition, many applied studies still rely primarily on I-association. For instance, Scher et al. [ 5 ] evaluated the individual-level surrogacy of circulating tumor cells for survival in prostate cancer; Fokas et al. [ 6 ] investigated the individual-level surrogacy of the neoadjuvant rectal score for disease-free survival in rectal cancer; and Yee et al. [ 7 ] attempted to identify pathologic complete response as a surrogate for survival endpoints in breast cancer based solely on strong I-association. These approaches do not follow FDA guidelines and thus the biomarkers are not considered valid surrogate endpoints. A possible reason for the limited evaluation of T-association is that existing statistical methods may be complex and difficult to interpret [ 4 , 8 ]. As a result, some recent work has adopted simpler approaches; for example, Ye et al. [ 9 ] evaluated T-association using the Spearman correlation between estimated treatment effects, an approach that does not account for the within-study uncertainty of the treatment effect estimates. To facilitate a more rigorous evaluation of T-association, we propose a new method based on the bivariate random-effects meta-analysis (BRMA) model [ 10 ], which can be used to synthesize two correlated endpoints. The model jointly accounts for both within- study and between-study variances in the estimated treatment effects on the biomarker and the true endpoint, allowing individual weighting of each study. Both frequentist and Bayesian frameworks are applied for estimation and inference. By quantifying T- association through the estimated correlation, our approach provides a practical and interpretable framework that aligns with FDA guidance for surrogate endpoint validation. This article is organized as follows. Section 2 introduces the proposed model and details both frequentist and Bayesian approaches for parameter estimation. Section 3 presents simulation studies and two real-world applications. Finally, Section 4 discusses the findings. 2 Methods 2.1 Model Setup For each study i = 1, …, n , we let y i 1 and y i 2 denote the statistics quantifying the treatment effect on the biomarker of interest and the true endpoint, respectively. For example, y i 1 and y i 2 are the estimates of log(OR) and log(HR), which reflect the treatment effects on dichotomous biomarkers and time-to-event endpoints. Here, OR and HR refer to odds ratio and hazard ratio, respectively. When the number of participants within study i is large, both y i 1 and y i 2 are typically assumed to approximate a normal distribution. To evaluate T-association, we apply the BRMA model proposed by Riley et al. [ 10 ] to synthesize the two treatment effect estimates. The BRMA model can be expressed as follows: where y i = ( y i 1 , y i 2 ) ′ and β = ( β 1 , β 2 ) ′ , in which the prime symbol (·) ′ denotes the transpose of a vector or matrix. The parameters β 1 and β 2 represent the mean treatment effects for the biomarker and the true endpoint, respectively. The error term e i is assumed to follow a normal distribution with the covariance Φ i . Within Φ i , the parameters and capture the between-study variances, while and are the known within-study variances, which are typically reported in study i . The correlation parameter ρ is the key evaluation metric of T-association. This model offers two advantages. First, it accounts for between- and within-study variances, allowing each study to be appropriately weighted based on its precision through and . Second, it captures the overall correlation between treatment effects using a single parameter ρ , rather than decomposing it into between- and within-study correlations, which is practical given that within-study correlations are rarely available. In the following sections, we present frequentist and Bayesian approaches for obtaining point and interval estimates for the parameters of the BRMA model. 2.2 Frequentist Approach Under the frequentist framework, maximum likelihood (ML) and restricted maximum likelihood (REML) methods are commonly used to estimate model parameters. In the following, we detail the procedures for obtaining ML and REML estimates, along with their associated confidence intervals (CI). Let and , where s i = ( s i 1 , s i 2 ) ′ , i = 1, …, n . Maximum likelihood Let , where β = ( β 1 , β 2 ) ′ . Given the observed data y and s , the full likelihood function can be expressed as: The corresponding log-likelihood function is: where X = ( I 2 , …, I 2 ) ′ with I 2 denoting the 2 by 2 identity matrix, and Φ is a block- diagonal matrix with each study’s covariance Φ i along the diagonal, i.e., Φ = diag(Φ 1 , …, Φ n ). The ML estimate of is obtained by solving U ( θ ) = ∂ℓ ( θ | y, s ) /∂ θ = 0 . The detailed expression for the score function U ( θ ) is provided in the Supplementary Materials. Since no closed-form solution exists for the equation, the ML estimate is obtained iteratively as follows: Step 1 Set the initial parameter value as . To ensure numerical stability and enforce parameter constraints (positive variances and | ρ | < 1), the variance components are reparameterized on the log scale, and the correlation is reparameterized using the Fisher z -transformation, which is defined as Step 2 Solve the system of score equations U ( θ ) = 0 using the Newton–Raphson algorithm, starting from the initial value θ (0) . Continue the iterations, indexed by m , until convergence, which is defined by ∥ θ ( m ) − θ ( m −1) ∥ 0. In our implementation, the score equations are solved by using the nleqslv function from the nleqslv package in R . Restricted maximum likelihood Under the REML framework, the restricted log-likelihood function is obtained from integrating out β from L ( θ | y, s ), which yields unbiased estimates for variance components. Let , then the restricted log-likelihood is given by: where X and Φ are defined as in the above section, and β R = ( X ′ Φ −1 X ) −1 X ′ Φ −1 y . Similar to the ML approach, we obtain the REML estimate of by solving U R ( θ R ) = ∂ℓ R ( θ R | y, s ) /∂ θ R = 0 using the Newton-Raphson algorithm. The explicit expression for the score function U R ( θ R ) is detailed in the Supplemental Materials. Finally, once is obtained, the estimate for β is given by , where . Confidence intervals To quantify the uncertainty of the estimators and facilitate statistical inference, we construct CIs for all model parameters. Under standard regularity conditions, the ML and REML estimators are asymptotically normal, with variances estimated from the inverses of their observed information matrices, denoted by ℐ and ℐ R : For the REML estimator , the variance is estimated by . Detailed expressions for the observed information matrices are provided in the Supplemental Materials. CIs are then derived based on the asymptotic normality of the estimators. As mentioned previously, the variance and correlation parameters were reparameterized when solving the score functions. Accordingly, CIs for β are computed on the original scale, while CIs for the variance components , and the correlation ρ are first formed on transformed scales to respect parameter constraints and then back-transformed. Specifically, CIs for variance components are computed on the log scale: and then exponentiated to return to the original scale, where z 1− α/ 2 is the standard normal quantile and α is the pre-specified significance level. CI for the correlation is computed on the Fisher-transformed scale: and back-transformed using the hyperbolic tangent. The variance estimates for the estimates of the transformed parameters under ML and REML: and are obtained from ( J ′ ℐ J ) −1 and , respectively, where and . In addition to these large-sample normal-based intervals, we also apply a parametric bootstrap procedure to construct CIs for the model parameters under both ML and REML frameworks. 2.3 Bayesian Approach In real-world applications, the number of studies available to evaluate T-association may be limited. As a more robust alternative in limited study scenarios, we adopt Bayesian methods. Given the observed data ( y, s ) and a prior distribution π ( θ ) for the parameters, the posterior distribution of θ is expressed as where the full likelihood function L ( θ | y, s ) is defined in the maximum likelihood section. In this study, we consider the following priors: where IG ( a, b ) is the inverse-gamma distribution with shape parameter a and scale parameter b , and U (−1, 1) is the uniform distribution over (−1, 1). To ensure weak informativeness, we choose a large value for σ 2 and small values for both a and b . The uniform prior on ρ is commonly used to reflect a flat, noninformative distribution over the feasible correlation range. Although the inverse-gamma priors with small shape and scale parameters are often used as noninformative priors for variance components, Gelman [ 11 ] criticized them for placing excessive mass near zero. As an alternative, they suggested a half- t prior, a folded Student’s- t distribution truncated to be positive, for ψ j . Gelman [ 11 ] showed that this prior provides better regularization and more stable inference for variance parameters. Accordingly, we also consider: where half- t ( ν, τ ) denotes a half- t distribution with degrees of freedom ν and scale parameter τ . For the correlation, in addition to the uniform prior, we also consider a Normal prior on the Fisher-transformed correlation: which induces a mild shrinking effect toward zero and serves as a weakly informative prior when is appropriately chosen. Thus, four prior combinations are provided: 2 priors for variance components and 2 priors for correlation. Since the resulting posterior distributions do not have closed-form expressions, we employ Markov Chain Monte Carlo (MCMC) to sample from the implicit posterior distributions. Specifically, we adopt Hamiltonian Monte Carlo (HMC) [ 12 ], a modern and efficient technique that leverages gradient information from the posterior to propose distant, high-probability values. Compared to traditional random-walk-based methods like Metropolis-Hastings, HMC significantly improves sampling efficiency, especially in settings where parameters exhibit strong correlations [ 13 ]. We implement HMC using the cmdstan model function from the cmdstanr package in R [ 14 ]. The function uses the No-U-Turn Sampler, an adaptive variant of HMC that automatically tunes the trajectory length to improve sampling efficiency [ 13 ]. The posterior distribution is constructed using four parallel Markov chains. Each chain is run for 1,000 warm-up iterations followed by 1,000 sampling iterations, yielding a total of 4,000 post–warm-up draws. Posterior means of the parameters θ , including the key T- association metric ρ , are used as point estimates. The 100(1 − α )% credible intervals (CrI) are constructed using the α/ 2 and 1 − α/ 2 quantiles of the corresponding marginal posterior distributions, where α is the pre-specified credible level. 3 Numerical Studies 3.1 Simulation Studies We conducted simulation studies to assess the performance of the proposed methods. In the simulations, the treatment effects for the biomarker and the true endpoint were generated from model (1) with parameters β ′ = (0.6, −0.2) and . These values were selected to reflect realistic treatment effect magnitudes and between-trial variability observed in real data [ 9 ]. Moreover, to investigate the impact of correlation, the parameter ρ was set to 0.1, 0.4, and 0.7, corresponding to low, moderate, and high correlations. Within-study standard errors s i 1 and s i 2 , assumed to be known, were generated from a uniform distribution U (0.05, 0.5). The number of trials we consider include: n = 6, 10, and 15. For each scenario, 1,000 simulation replicates were performed, and estimation results from both frequentist and Bayesian approaches are reported. Frequentist approach Parameter estimates were obtained using both ML and REML methods. Estimation performance was evaluated based on empirical bias (Bias), empirical standard error (SE em ) calculated from 1,000 simulation replicates, the average theoretical standard error (SE th ) across 1,000 replicates, and the coverage probability of the 95% CI (CP th ) based on the theoretical standard error. In addition, the coverage probability of the 95% CI obtained via parametric bootstrap (CP b ) was calculated using 1,000 bootstrap samples. Results are presented in Table 1 . View this table: View inline View popup Download powerpoint Table 1. Simulation results under the frequentist framework across different numbers of trials ( n ) and true correlations ( ρ true ). The significance level α was set to 0.05. “Par” stands for parameter. For the ML estimate of ρ , bias is small and consistent across various values of ρ and decreases as n increases, ranging from -0.003 to 0.035. Both SE th and SE em decrease with increasing ρ and n . Overall, SE th is smaller than SE em , with the discrepancy more pronounced for smaller sample sizes ( n = 6 and 10). When n increases to 15, SE th aligns more closely with SE em . CP th shows a slight increasing trend with ρ , while CP b does not follow the same pattern. Both CP th and CP b are smaller than the nominal 95% level but improve with larger n . REML results are similar, except that SE th is larger, coverage probabilities are generally larger, and CP th and CP b are more consistent. For the ML estimate of β , bias is small and stable across values of ρ and decreases with n , ranging from -0.010 to 0.002. SE th and SE em also show little variation with ρ . Both SE metrics decrease as n increases, with SE th slightly smaller than SE em . CP th and CP b are below the 95% level but CP th is better than CP b ; both improve with larger n . REML results are similar, except that SE th is larger, coverage probabilities perform better, and CP th and CP b are more aligned. For the ML estimate of , bias is small and stable across ρ and decreases with n , ranging from -0.039 to 0.014. SE th and SE em also show little variation with ρ and decrease as n increases. Notably, SE th is larger than SE em at n = 6. CP th is larger than the 95% level but improves with larger n , while CP b is smaller than the 95% level and has no substantial improvement with n . REML results are similar, except both SE metrics are larger and coverage probabilities perform better. In summary, both ML and REML estimates exhibit small biases, and both biases and standard errors decrease with sample sizes. Coverage probabilities improve with sample sizes, except CP b for variance components. REML generally outperforms ML in terms of coverage probability. Bayesian approach We examined several prior combinations to evaluate estimation performance. The prior for β 1 and β 2 were set as N (0, 2.5 2 ), and four prior combinations were considered for variance and correlation parameters: Posterior sampling was performed using the HMC method, drawing 4,000 samples for each parameter. Table 2 presents bias and empirical coverage (EC) for each parameter. Across 4 prior combinations, bias in the ρ estimate increases as ρ increases. EC for ρ is close to 95% when ρ = 0.1 and 0.4, but slightly below 95% when ρ = 0.7. Bias decreases and EC improves as n increases. Comparing different priors on ρ , the Normal prior (Priors 2 and 4) slightly reduces bias and improves EC compared to the U (−1, 1) prior (Priors 1 and 3). View this table: View inline View popup Download powerpoint Table 2. Simulation results under the Bayesian framework across different numbers of trials ( n ) and true correlations ( ρ true ). Priors 1–4 correspond to different combinations of variance priors, either IG(0.001, 0.001) or half- t (3, 0.5); and correlation priors, either U (−1, 1) or F ρ ∼ N (0, 1). For and , bias decreases as ρ increases, and EC is close to 95% although it is less stable for small n = 6 and 10. Bias decreases and EC improves with larger n . Among all priors on , the half- t (3, 0.3) prior (Priors 3 and 4) yields smaller bias and better EC than the IG (0.001, 0.001) prior (Priors 1 and 2), particularly for small sample sizes. For β 1 and β 2 , estimates exhibit small biases and reasonable ECs, which are not largely affected by priors on ρ , , and . Overall, Prior 4 demonstrates the most robust performance across all parameters. Compared to the frequentist approach, Bayesian estimates show slightly larger biases, likely due to the use of flat or weakly informative priors. Nevertheless, EC is closer to the nominal level of 0.95, indicating that the Bayesian approach provides more reliable interval estimates. 3.2 Real Application To demonstrate the proposed method in a real-world setting, we analyzed data from 14 non–small cell lung cancer (NSCLC) trials comparing chemotherapy and immunotherapy [ 15 – 27 ]. We considered two scenarios: the first evaluated the T-association between progression-free survival (PFS) and OS, and the second evaluated the T-association between objective response and OS. Based on the simulation results, the REML approach was selected for the frequentist analysis, and Prior 4 was used for the Bayesian analysis, as it demonstrated the most robust performance across parameters. Scenario 1 All 14 NSCLC trials reported hazard ratios for PFS and OS, which were used to examine the T-association between these endpoints. Figure 2(a) visualizes the log(HR) estimates for PFS and OS across the 14 trials. The half-widths and half-heights of the ellipses represent 1.96 times the within-trial standard errors. Download figure Open in new tab Figure 2. Visualization of transformed treatment effect estimates (solid points) for the two scenarios. The half-widths and half-heights of the ellipses represent 1.96 times the within-trial standard errors. (a) log(HR) for OS versus log(HR) for PFS across 14 trials in Scenario 1. (b) log(HR) for OS versus log(OR) for objective response across 6 trials in Scenario 2. Using the proposed method, the estimated correlation ( ρ ) between the log(HR) estimates on PFS and OS was 0.71 (95% bootstrapped CI: 0.36 to 0.91; 95% normal-based CI: 0.32 to 0.90) using the REML approach; and 0.63 (95% CrI: 0.23 to 0.88) using the Bayesian approach ( Table 3 ). These results show a strong correlation between treatment effects on PFS and OS, offering supportive evidence for FDA consideration regarding the validity of PFS as a surrogate endpoint for OS in terms of T-association. For a reference, the Spearman correlation, which ignores within-study variability, was 0.76 (95% bootstrapped CI: 0.33 to 0.95), showing a higher correlation estimate and a wider interval than REML. This wider interval is probably caused by the ignorance of within-trial variances. View this table: View inline View popup Download powerpoint Table 3. Estimates of the model parameters with 95% confidence/credible intervals for two real-world scenarios. Scenario 2 Among the 14 NSCLC trials, 6 trials reported odds ratios for objective response comparing chemotherapy and immunotherapy, allowing assessment of the T-association between objective response and OS. Objective response, defined as a complete or partial tumor response, is a potential surrogate endpoint that can be observed earlier than OS. Figure 2 (b) displays the transformed treatment effect estimates from the six trials. Using the proposed method, the estimated correlation between treatment effects on objective response and OS was -0.79 (95% bootstrapped CI: -0.98 to -0.16; 95% normal- based CI: -0.96 to -0.16) using the REML approach ( Table 3 ). The Bayesian approach also showed a negative correlation of -0.56, but with a wider and more conservative credible interval (95% CrI: -0.90 to -0.07). These results provide evidence that the objective response may serve as a surrogate endpoint for OS. For a reference, the Spearman correlation was -0.77 (bootstrapped CI unavailable due to the limited number of trials). 4 Discussion According to FDA guidelines, a biomarker is considered a valid surrogate if it satisfies two key criteria: I-association and T-association. I-association is commonly evaluated, but T-association is often overlooked, likely due to the absence of suitable analysis methods. To fill this gap, we propose a new analysis method based on the BRMA model to evaluate T-association. Parameter estimation is performed using both frequentist and Bayesian approaches. We believe the proposed method will serve as the statistical foundation for future FDA Accelerated Approval drugs. Our simulation results indicate that, compared to REML, the ML approach is less stable, sometimes yielding narrower CIs and smaller coverage probability for the key correlation. Therefore, when the number of studies is limited, the REML approach is preferable. In addition, for small sample sizes, large-sample normal-based coverage probabilities for variance components under both ML and REML tend to be overly large due to large standard errors and wide CIs. When fewer than 10 studies are available, we recommend constructing CIs for variance components using the parametric bootstrap method under REML when applying a frequentist approach. Compared to the frequentist approach, the Bayesian approach provides more reliable interval estimates although its point estimates may have larger bias due to the influence of priors. Simulation results suggest that the half- t ( ν, τ ) prior on ψ j is more robust than the inverse-gamma IG ( a, b ) prior on , and the Normal prior on the Fisher- z transformed correlation is preferable to the uniform U (−1, 1) prior on ρ . Specifying a prior on the Fisher- z transformed correlation allows sampling on the unconstrained real line rather than within a bounded interval, which improves numerical stability and MCMC efficiency, as suggested in reference [ 14 ]. The number of trials available to evaluate T-association is often limited, raising the question of the minimum number of trials required to obtain reliable correlation estimates. Based on our simulation results, we recommend including at least six trials. Under the frequentist framework, the ML method estimates five parameters, while the REML method estimates three. When the true correlation is moderate (e.g., ρ = 0.4), using only five trials led to a high proportion of divergent estimates across 1,000 simulations: 57.4% for ML and 44% for REML. Although the Bayesian approach does not encounter such numerical divergence, its performance at very small sample sizes is also unstable. The bivariate normal assumption in the BRMA model may be restrictive, although the observed treatment effects on the biomarker of interest and the true endpoint are generally assumed to individually follow a normal distribution. To assess the validity of this assumption, we introduce a simple diagnostic procedure: applying the Shapiro–Wilk test to check whether the standardized residuals approximate the standard normal distribution. In the real-data application, the Shapiro–Wilk test on the standardized residuals yielded p -values of 0.25 and 0.30 for the two scenarios, providing no evidence against normality. Clinical trial designs can differ in various aspects, such as treatment regimen, leading to heterogeneity across studies. The proposed method can theoretically be extended to account for such heterogeneity by incorporating trial-level covariates. Specifically, the parameter β in the BRMA model can be replaced with , where X i is the covariate matrix for study i , and β c is the corresponding coefficient vector. However, due to the limited number of trials available in practice, we do not perform this extension in the current study. Data Availability All data produced in the present work are contained in the manuscript Funding Information This work is supported by the National Institutes of Health [P30 CA068485, U2C CA233291, R01 CA252964, and U54 CA260560, CA163072]. Supplementary Material Online supplemental material. The freely available R package at https://github.com/jybelindahung/T-association . Conflict of Interest Statement The authors have declared no conflict of interest. Footnotes This version has been extensively revised across all sections to present a complete manuscript, with updates to the introduction, methods, results, simulations, and discussion. References [1]. ↵ US Food and Drug Administration . Accelerated approval–expedited program for serious conditions . 2024 . https://www.fda.gov/media/184120/download . Accessed October 13, 2025 . [2]. ↵ US Food and Drug Administration . CDER drug and biologic accelerated approvals based on a surrogate endpoint. 2025 . https://www.fda.gov/media/151146/download?attachment . Accessed October 13, 2025 . [3]. ↵ US Food and Drug Administration . Hematologic malignancies: regulatory considerations for use of minimal residual disease in development of drug and biological products for treatment . 2020 . https://www.fda.gov/media/134605/download . Accessed October 13, 2025 . [4]. ↵ Buyse M , Molenberghs G , Burzykowski T , Renard D , Geys H. The validation of surrogate endpoints in meta-analyses of randomized experiments . Biostatistics . 2000 ; 1 ( 1 ): 49 – 67 . OpenUrl CrossRef PubMed [5]. ↵ Scher HI , Heller G , Molina A , et al. Circulating tumor cell biomarker panel as an individual-level surrogate for survival in metastatic castration-resistant prostate cancer . Journal of Clinical Oncology . 2015 ; 33 ( 12 ): 1348 – 1355 . OpenUrl Abstract / FREE Full Text [6]. ↵ Fokas E , Fietkau R , Hartmann A , et al. Neoadjuvant rectal score as individual-level surrogate for disease-free survival in rectal cancer in the CAO/ARO/AIO-04 randomized phase III trial . Annals of Oncology . 2018 ; 29 ( 7 ): 1521 – 1527 . OpenUrl PubMed [7]. ↵ Yee D , DeMichele AM , Yau C , et al. Association of event-free and distant recurrence– free survival with individual-level pathologic complete response in neoadjuvant treatment of stages 2 and 3 breast cancer: three-year follow-up analysis for the I-SPY2 adaptively randomized clinical trial . JAMA Oncology . 2020 ; 6 ( 9 ): 1355 – 1362 . OpenUrl PubMed [8]. ↵ Daniels MJ , Hughes MD . Meta-analysis for the evaluation of potential surrogate markers . Statistics in Medicine . 1997 ; 16 ( 17 ): 1965 – 1982 . OpenUrl CrossRef PubMed Web of Science [9]. ↵ Ye J , Ji X , Dennis PA , Abdullah H , Mukhopadhyay P. Relationship between progression-free survival, objective response rate, and overall survival in clinical trials of PD-1/PD-L1 immune checkpoint blockade: a meta-analysis . Clinical Pharmacology & Therapeutics . 2020 ; 108 ( 6 ): 1274 – 1288 . OpenUrl PubMed [10]. ↵ Riley RD , Thompson JR , Abrams KR . An alternative model for bivariate randomeffects meta-analysis when the within-study correlations are unknown . Biostatistics . 2008 ; 9 ( 1 ): 172 – 186 . OpenUrl CrossRef PubMed Web of Science [11]. ↵ Gelman A. Prior distributions for variance parameters in hierarchical models (comment on article by Browne and Draper) . Bayesian Analysis . 2006 ; 1 ( 3 ): 515 – 534 . OpenUrl [12]. ↵ Brooks S , Gelman A , Jones GL , Meng XL Neal RM . MCMC using Hamiltonian dynamics . In: Brooks S , Gelman A , Jones GL , Meng XL , eds. Handbook of Markov Chain Monte Carlo . Chapman & Hall/CRC ; 2011 : 113 – 162 . [13]. ↵ Hoffman MD , Gelman A. The No-U-Turn sampler: adaptively setting path lengths in Hamiltonian Monte Carlo . Journal of Machine Learning Research . 2014 ; 15 ( 1 ): 1593 – 1623 . OpenUrl [14]. ↵ Stan Development Team . Stan modeling language user’s guide and reference manual, version 2.37 . 2025 . https://mc-stan.org/docs/reference-manual/ . Accessed October 19, 2025 . [15]. ↵ Carbone DP , Reck M , Paz-Ares L , et al. First-line nivolumab in stage IV or recurrent non–small-cell lung cancer . New England Journal of Medicine . 2017 ; 376 ( 25 ): 2415 – 2426 . OpenUrl CrossRef PubMed [16]. Rizvi NA , Cho BC , Reinmuth N , et al. Durvalumab with or without tremelimumab vs standard chemotherapy in first-line treatment of metastatic non–small cell lung cancer: the MYSTIC phase 3 randomized clinical trial . JAMA Oncology . 2020 ; 6 ( 5 ): 661 – 674 . OpenUrl PubMed [17]. Reck M , Rodríguez-Abreu D , Robinson AG , et al. Updated analysis of KEYNOTE024: pembrolizumab versus platinum-based chemotherapy for advanced non–smallcell lung cancer with PD-L1 tumor proportion score of 50% or greater . Journal of Clinical Oncology . 2019 ; 37 ( 7 ): 537 – 546 . OpenUrl CrossRef PubMed [18]. Mok TSK , Wu YL , Kudaba I , et al. Pembrolizumab versus chemotherapy for previously untreated, PD-L1–expressing, locally advanced or metastatic non–small-cell lung cancer (KEYNOTE-042): a randomised, open-label, controlled, phase 3 trial . The Lancet . 2019 ; 393 ( 10183 ): 1819 – 1830 . OpenUrl [19]. Hellmann MD , Paz-Ares L , Bernabe Caro R , et al. Nivolumab plus ipilimumab in advanced non–small-cell lung cancer . New England Journal of Medicine . 2019 ; 381 ( 21 ): 2020 – 2031 . OpenUrl CrossRef PubMed [20]. Spigel D , De Marinis F , Giaccone G , et al. IMpower110: interim overall survival analysis of a phase III study of atezolizumab vs platinum-based chemotherapy as first-line treatment in PD-L1–selected NSCLC . Annals of Oncology . 2019 ; 30 : v915 . OpenUrl CrossRef [21]. Barlesi F , Vansteenkiste J , Spigel D , et al. Avelumab versus docetaxel in patients with platinum-treated advanced non–small-cell lung cancer (JAVELIN Lung 200): an open-label, randomised, phase 3 study . The Lancet Oncology . 2018 ; 19 ( 11 ): 1468 – 1479 . OpenUrl CrossRef PubMed [22]. Fehrenbacher L , von Pawel J , Park K , et al. Updated efficacy analysis including secondary population results for OAK: a randomized phase III study of atezolizumab versus docetaxel in patients with previously treated advanced non–small cell lung cancer . Journal of Thoracic Oncology . 2018 ; 13 ( 8 ): 1156 – 1170 . OpenUrl PubMed [23]. Fehrenbacher L , Spira A , Ballinger M , et al. Atezolizumab versus docetaxel for patients with previously treated non–small-cell lung cancer (POPLAR): a multicentre, open-label, phase 2 randomised controlled trial . The Lancet . 2016 ; 387 ( 10030 ): 1837 – 1846 . OpenUrl [24]. Planchard D , Reinmuth N , Orlov S , et al. ARCTIC: durvalumab with or without tremelimumab as third-line or later treatment of metastatic non–small-cell lung cancer . Annals of Oncology . 2020 ; 31 ( 5 ): 609 – 618 . OpenUrl CrossRef PubMed [25]. Brahmer J , Reckamp KL , Baas P , et al. Nivolumab versus docetaxel in advanced squamous-cell non–small-cell lung cancer . New England Journal of Medicine . 2015 ; 373 ( 2 ): 123 – 135 . OpenUrl CrossRef PubMed [26]. Borghaei H , Paz-Ares L , Horn L , et al. Nivolumab versus docetaxel in advanced nonsquamous non–small-cell lung cancer . New England Journal of Medicine . 2015 ; 373 ( 17 ): 1627 – 1639 . OpenUrl CrossRef PubMed [27]. ↵ Herbst RS , Baas P , Kim DW , et al. Pembrolizumab versus docetaxel for previously treated, PD-L1–positive, advanced non–small-cell lung cancer (KEYNOTE-010): a randomised controlled trial . The Lancet . 2016 ; 387 ( 10027 ): 1540 – 1550 . OpenUrl View the discussion thread. Back to top Previous Next Posted February 02, 2026. Download PDF Supplementary Material Data/Code Email Thank you for your interest in spreading the word about medRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following Evaluation Methods for T-association of a Surrogate Endpoint Message Subject (Your Name) has forwarded a page to you from medRxiv Message Body (Your Name) thought you would like to see this page from the medRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share Evaluation Methods for T-association of a Surrogate Endpoint Jo-Ying Hung , Chih-Yuan Hsu , Pei-Fang Su , Yu Shyr medRxiv 2025.08.28.25334653; doi: https://doi.org/10.1101/2025.08.28.25334653 Share This Article: Copy Citation Tools Evaluation Methods for T-association of a Surrogate Endpoint Jo-Ying Hung , Chih-Yuan Hsu , Pei-Fang Su , Yu Shyr medRxiv 2025.08.28.25334653; doi: https://doi.org/10.1101/2025.08.28.25334653 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Health Policy Subject Areas All Articles Addiction Medicine (567) Allergy and Immunology (863) Anesthesia (297) Cardiovascular Medicine (4411) Dentistry and Oral Medicine (443) Dermatology (380) Emergency Medicine (606) Endocrinology (including Diabetes Mellitus and Metabolic Disease) (1505) Epidemiology (15205) Forensic Medicine (30) Gastroenterology (1119) Genetic and Genomic Medicine (6574) Geriatric Medicine (666) Health Economics (994) Health Informatics (4511) Health Policy (1365) Health Systems and Quality Improvement (1608) Hematology (537) HIV/AIDS (1263) Infectious Diseases (except HIV/AIDS) (15903) Intensive Care and Critical Care Medicine (1103) Medical Education (620) Medical Ethics (144) Nephrology (666) Neurology (6573) Nursing (345) Nutrition (998) Obstetrics and Gynecology (1139) Occupational and Environmental Health (954) Oncology (3319) Ophthalmology (968) Orthopedics (369) Otolaryngology (420) Pain Medicine (435) Palliative Medicine (129) Pathology (662) Pediatrics (1689) Pharmacology and Therapeutics (691) Primary Care Research (710) Psychiatry and Clinical Psychology (5422) Public and Global Health (9205) Radiology and Imaging (2191) Rehabilitation Medicine and Physical Therapy (1367) Respiratory Medicine (1191) Rheumatology (593) Sexual and Reproductive Health (709) Sports Medicine (529) Surgery (709) Toxicology (99) Transplantation (288) Urology (265) (function(){function c(){var b=a.contentDocument||a.contentWindow.document;if(b){var d=b.createElement('script');d.innerHTML="window.__CF$cv$params={r:'9fe9f97eac06e2c5',t:'MTc3OTI2NTc3Ng=='};var a=document.createElement('script');a.src='/cdn-cgi/challenge-platform/scripts/jsd/main.js';document.getElementsByTagName('head')[0].appendChild(a);";b.getElementsByTagName('head')[0].appendChild(d)}}if(document.body){var a=document.createElement('iframe');a.height=1;a.width=1;a.style.position='absolute';a.style.top=0;a.style.left=0;a.style.border='none';a.style.visibility='hidden';document.body.appendChild(a);if('loading'!==document.readyState)c();else if(window.addEventListener)document.addEventListener('DOMContentLoaded',c);else{var e=document.onreadystatechange||function(){};document.onreadystatechange=function(b){e(b);'loading'!==document.readyState&&(document.onreadystatechange=e,c())}}}})();
Text is read by the "Ask this paper" AI Q&A widget below.
Extraction quality varies by source — PMC NXML preserves structure
cleanly, OA-HTML may include some navigation residue, and OA-PDF can
have broken hyphenation. The publisher copy
(via DOI)
is the canonical version.