Full text
55,449 characters
· extracted from
preprint-html
· click to expand
Automated Quantification of Decreased FAF in Stargardt Disease: Validation of a Novel Method Compared to Manual Grading Standards | medRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-P4HH5NV'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search Automated Quantification of Decreased FAF in Stargardt Disease: Validation of a Novel Method Compared to Manual Grading Standards Mohamed I. Ahmed , Hikmet Yucel , Rubbia Afridi , Thales A.C. de Guimarães , Isabel Sendino-Tenorio , Nam V. Nguyen , Kiran Romaisa , Sidrah Khan , Ufaq Khan , Syeda Sharaf un Nisa , Mauro Campigotto , Amir Hariri , Michel Michaelides , Hendrik P.N. Scholl , Nathan Mata , Quan D. Nguyen , Yasir J. Sepah doi: https://doi.org/10.1101/2025.07.07.25330927 Mohamed I. Ahmed 1 Ocular Imaging Research and Reading Center (OIRRC) , Sunnyvale, CA, United States MD, MSc Find this author on Google Scholar Find this author on PubMed Search for this author on this site Hikmet Yucel 1 Ocular Imaging Research and Reading Center (OIRRC) , Sunnyvale, CA, United States PhD Find this author on Google Scholar Find this author on PubMed Search for this author on this site Rubbia Afridi 1 Ocular Imaging Research and Reading Center (OIRRC) , Sunnyvale, CA, United States MBBS, MPH Find this author on Google Scholar Find this author on PubMed Search for this author on this site Thales A.C. de Guimarães 1 Ocular Imaging Research and Reading Center (OIRRC) , Sunnyvale, CA, United States 2 Department of Ophthalmology, Faculdade São Leopoldo Mandic , Campinas, SP, Brazil MD, PhD Find this author on Google Scholar Find this author on PubMed Search for this author on this site Isabel Sendino-Tenorio 1 Ocular Imaging Research and Reading Center (OIRRC) , Sunnyvale, CA, United States 3 Department of Ophthalmology, University Hospital of Leon, Leon, Spain; University of Leon , Leon, Spain MD Find this author on Google Scholar Find this author on PubMed Search for this author on this site Nam V. Nguyen 1 Ocular Imaging Research and Reading Center (OIRRC) , Sunnyvale, CA, United States BS Find this author on Google Scholar Find this author on PubMed Search for this author on this site Kiran Romaisa 1 Ocular Imaging Research and Reading Center (OIRRC) , Sunnyvale, CA, United States MBBS Find this author on Google Scholar Find this author on PubMed Search for this author on this site Sidrah Khan 1 Ocular Imaging Research and Reading Center (OIRRC) , Sunnyvale, CA, United States PharmD Find this author on Google Scholar Find this author on PubMed Search for this author on this site Ufaq Khan 1 Ocular Imaging Research and Reading Center (OIRRC) , Sunnyvale, CA, United States BSc Find this author on Google Scholar Find this author on PubMed Search for this author on this site Syeda Sharaf un Nisa 1 Ocular Imaging Research and Reading Center (OIRRC) , Sunnyvale, CA, United States PharmD, MPhil Find this author on Google Scholar Find this author on PubMed Search for this author on this site Mauro Campigotto 1 Ocular Imaging Research and Reading Center (OIRRC) , Sunnyvale, CA, United States BSc, MSEE Find this author on Google Scholar Find this author on PubMed Search for this author on this site Amir Hariri 1 Ocular Imaging Research and Reading Center (OIRRC) , Sunnyvale, CA, United States MD Find this author on Google Scholar Find this author on PubMed Search for this author on this site Michel Michaelides 4 UCL Institute of Ophthalmology, University College London , London, UK 5 Moorfields Eye Hospital NHS Foundation Trust , London, UK BSc, MRCOphth Find this author on Google Scholar Find this author on PubMed Search for this author on this site Hendrik P.N. Scholl 6 Belite Bio Inc , San Diego, CA, United States 7 Medical University of Vienna, Department of Clinical Pharmacology , Vienna, Austria 8 Pallas Kliniken AG, Pallas Klinik Zürich , Zürich, Switzerland 9 European Vision Institute , Basel, Switzerland MD, MA Find this author on Google Scholar Find this author on PubMed Search for this author on this site Nathan Mata 6 Belite Bio Inc , San Diego, CA, United States PhD Find this author on Google Scholar Find this author on PubMed Search for this author on this site Quan D. Nguyen 10 Ophthalmology, Stanford University School of Medicine , Palo Alto, CA, United States MD, MSc Find this author on Google Scholar Find this author on PubMed Search for this author on this site Yasir J. Sepah 10 Ophthalmology, Stanford University School of Medicine , Palo Alto, CA, United States MBBS Find this author on Google Scholar Find this author on PubMed Search for this author on this site For correspondence: yjs{at}stanford.edu Abstract Full Text Info/History Metrics Data/Code Preview PDF Abstract Purpose To evaluate the repeatability and reproducibility of a novel automated method compared with manual segmentation for measuring decreased autofluorescence (DAF) and definitely decreased autofluorescence (DDAF) in fundus autofluorescence (FAF) images of patients with Stargardt disease. Design Cross-sectional reproducibility and agreement study. Participants A total of 316 eyes from 158 genetically confirmed Stargardt patients were analyzed. For intra-grader repeatability, 114 FAF images were reassessed in a masked, repeated-measures design. Methods DAF and DDAF lesion areas were independently quantified by five certified graders using either manual delineation with Heidelberg RegionFinder or a threshold-based automated algorithm. Agreement and repeatability were assessed using intraclass correlation coefficients (ICC), standard error of measurement (SEM), minimal detectable change (MDC), Lin’s concordance correlation coefficient (CCC), Bland–Altman plots, and Passing–Bablok regression. Both raw and square-root-transformed lesion areas were evaluated. Main Outcome Measures Repeatability (intra-grader ICC, SEM, MDC), reproducibility (inter-grader ICC), and agreement (CCC, bias in regression analysis) between and within manual and automated methods. Results The automated method achieved excellent intra-grader repeatability for both DAF and DDAF (ICCs ≥0.988, SEM ≤0.71 mm², MDC ≤1.98 mm²), with minimal operator influence. Manual measurements showed variable repeatability (DAF ICCs 0.909–0.974; DDAF ICCs as low as 0.837), with square-root transformation reducing SEM and MDC. Inter-grader reproducibility was highest for automated methods (ICC = 0.989–0.992), whereas manual methods ranged from 0.764–0.939 (raw) and 0.867–0.922 (transformed). Cross-method agreement was strong (CCC = 0.91–0.96), though minor proportional and constant bias was observed in raw DAF data. Conclusions The automated approach provides near-perfect repeatability and high agreement with manual grading, offering a scalable, objective alternative for quantifying hypo-autofluorescent lesions in Stargardt disease. Manual methods are generally reliable but more variable, especially for DDAF, and benefit from square-root transformation. Introduction Fundus autofluorescence (FAF) imaging is a critical tool for monitoring retinal degenerations, including Stargardt disease (STGD1), due to its capacity to visualize lipofuscin distribution in the retinal pigment epithelium (RPE) in vivo – frequently before more prominent changes appear in the fundus 1 . STGD1, caused by ABCA4 mutations, leads to lipofuscin accumulation and eventually to progressive RPE and photoreceptor atrophy. On FAF, affected areas typically appear as dark regions of decreased autofluorescence (DAF), which are often described as well-demarcated in the literature 2 . However, in clinical practice, lesion boundaries can vary considerably, and in many cases, the demarcation is indistinct, particularly at the lesion margins. Initial quantification of DAF relied on manual planimetry, as shown in a longitudinal study by Fujinami et al. (2013), which tracked atrophic area growth over nearly a decade 3 . Although effective, manual tracing is laborious and subject to reader variability. To standardize assessments, tools like Heidelberg’s RegionFinder introduced semiautomated segmentation using seed-and-grow algorithms and adjustable thresholds, but considerable human involvement remained necessary 4 . The ProgStar study, a multicenter effort led by Scholl and colleagues, formalized DAF measurement protocols for clinical trials 5 . FAF images were graded for “definitely decreased” (DDAF; ≥90% black relative to the optic disc) and “questionably decreased” (QDAF; 50–90%) autofluorescence. RegionFinder was used to segment lesions, with lesion area quantified for both DDAF and QDAF alone and DDAF+QDAF combined. This protocol improved consistency and became the standard for evaluating STGD1 progression 6 . Despite these advances, variability in lesion delineation remains a challenge. ProgStar addressed this through dual-reader workflows and adjudication of non-consensus cases. Still, subtle differences arise, especially at lesion borders or in differentiating DDAF from QDAF. Although Georgiou et al. (2020) reported excellent intraclass correlation coefficients (ICCs) of 0.995 (intra) and 0.987 (inter) for total DAF area, significant non-consensus was found in around 42% of cases, mostly due to disagreements on whether an area has 50-90% or 90-100% DAF leading the investigators to decide to only compare measurements of DAF 4 . Factors like image quality, illumination, and lesion edge gradation contribute to measurement variability. STGD1 progression varies with disease stage. In early stages, atrophy progresses slowly (∼0.05– 0.1 mm²/year), but in advanced stages, expansion can exceed 4 mm²/year 3 . ProgStar’s 12- and 24-month findings confirmed average lesion growth of ∼0.5–0.8 mm²/year 5 , 7 . Such small margins of change require higher precision in the measuring procedure. While these studies validated DAF as a reproducible metric, they also highlighted the burden of human grading. To address this, researchers have turned to automation. Early thresholding methods 8 lacked sensitivity for STGD1’s heterogeneity. More recently, Zhao et al. (2022) developed a deep learning model for FAF lesion segmentation, achieving high agreement with expert graders (Dice score ∼0.90 for DDAF). Performance for QDAF was lower (∼0.55), reflecting inherent ambiguity 9 . In summary, although FAF-based DAF quantification is central to STGD1 research, it is constrained by manual workflows and inter-reader variability. While emerging automated techniques offer promising solutions, this study aims to build on these efforts, proposing a novel, reproducible approach for quantifying DAF to support future clinical applications. Methods Study Design and Participants This observational study was conducted to assess the intra- and inter-grader repeatability and agreement of two lesion endpoints, Definitely Decreased Autofluorescence (DDAF) and overall Decreased Autofluorescence (DAF), using two distinct measurement approaches: a traditional semi-automated method based on a seed-and-growth algorithm (referred to as the manual method in this manuscript), and a novel automated method employing a threshold-based approach. The concordance between these two methods was evaluated to determine the degree of alignment between the automated system and established manual practices. A total of 316 eyes from 158 subjects diagnosed with Stargardt disease were retrospectively included. Data were curated from fundus autofluorescence (FAF) images acquired during the screening visits of the LBS-008-CT03 Clinical Trial (DRAGON) [ ClinicalTrials.gov ID: NCT05244304 ]. The study adhered to the tenets of the Declaration of Helsinki and received approval from the local institutional review board. Informed consent was obtained from all participants prior to inclusion. Measurement Protocol Five certified graders (RA, TG, IST, RK, NN) independently assessed the reduced autofluorescence measurements on the same set of images. Each grader performed a second round of measurements on 114 randomly selected images after a washout period of at least one week between sessions to minimize recall bias. Two distinct decreased autofluorescence endpoints were evaluated based on the definitions of the ProgStar study standard of DDAF and DAF (combined QDAF+DDAF) . For further analysis, each endpoint was also transformed using the square root function, yielding the transformed variables sqDAF and sqDDAF. This transformation, which is commonly employed in the analysis of similar lesion types such as geographic atrophy 10 , 11 , was applied to mitigate heteroscedasticity and reduce the influence of outliers, thereby improving the statistical robustness of subsequent reliability assessments. Measurement Methods Three graders (IST, RK, NN) performed manual delineation of lesion boundaries using a traditional semi-automated approach, referred to as the manual method in this manuscript (Region Finder, Heidelberg Engineering). This method relies on a seed-and-growth algorithm in which lesion boundaries are automatically generated by the software following grader input. Graders initiate the process by selecting a seed point within the atrophic area, and the software automatically delineates lesion boundaries, allowing manual adjustments to parameters such as “growth power” to refine segmentation. However, since graders retain full control over the thresholding parameters, particularly the “growth power,” this directly influences the extent of lesion boundary propagation. Two graders (RA, TG) employed a novel automated image analysis algorithm designed to enable high-throughput and reproducible quantification of autofluorescence lesion areas. The method requires the grader to manually identify the foveal center and the optic disc, as well as to provide anatomical reference points used to calibrate a uniform spatial scale. This calibration facilitates pixel-wise classification based on the characteristics of local neighborhoods relative to pre-specified, tunable intensity thresholds. The algorithm incorporates configurable parameters, including the ability to set a minimum lesion size threshold for inclusion in the aggregate lesion area, defined as ≥0.5 mm² in this study. It also allows restriction and stratifications of measurements to predefined regions of interest within the fundus. In this study, all measurements were limited to the central 6 mm of the fundus. The algorithm also supports vessel masking to exclude retinal vasculature from analysis, thereby minimizing the risk of overestimating lesion size. While the algorithm is capable of operating across multiple threshold levels, including those capturing the hyperautofluorescent spectrum, only two threshold levels, selected to reflect grading standards for Stargardt disease, were used in this study. Statistical Analysis All statistical analyses were conducted using industry-standard software, including IBM SPSS Statistics (Version 30). Descriptive statistics (mean, standard deviation, and range) were computed for each lesion endpoint (DAF and DDAF), both in their original form and after square-root transformation, to account for potential skewness and heteroscedasticity. Square-root transformation was applied to all DAF and DDAF values to stabilize variance and reduce skewness, as detailed in the Data Transformation and Rationale section. All reliability and agreement metrics are reported for both raw and transformed data. Intra-grader repeatability was assessed using the Intraclass Correlation Coefficient (ICC) under a two-way mixed-effects model with absolute agreement definition (single rater, single measurement). For each grader, ICCs were calculated separately for raw and transformed data. To quantify absolute measurement error, the Standard Error of Measurement (SEM) was derived using the formula: where pooled standard deviation was computed as: The Minimal Detectable Change (MDC), also referred to as the Coefficient of Repeatability (CR), was calculated as: Inter-grader repeatability (within-method reliability) was assessed using the same ICC framework for each pairwise comparison among graders operating under the same method (automated or manual). For each pair, both single-measure and average-measure ICCs were computed, along with SEM and MDC. The effect of square-root transformation was evaluated by comparing reliability metrics before and after transformation. Agreement between methods (automated vs. manual) was assessed using Lin’s Concordance Correlation Coefficient (CCC) to quantify overall concordance and Passing–Bablok regression to detect proportional and constant bias. CCC was computed manually using its standard definition, incorporating Pearson’s correlation, standard deviations, and mean differences. Passing–Bablok regression provided estimates of slope and intercept to assess systematic deviations from the identity line. For each endpoint and transformation state, the CCC, regression slope, and intercept were reported. Bland–Altman analysis was used to visualize intra- and inter-grader agreement within each method. Separate Bland–Altman plots were constructed for each pairwise comparison (raw and transformed), stratified by method and lesion type. To accommodate the number of comparisons (n = 20 for intra-grader and n = 16 for inter-grader), composite figures were generated to display representative plots for automated and manual graders across both lesion phenotypes. Throughout the analysis, special attention was given to evaluating the effect of square-root transformation on measurement precision and agreement. Comparisons were made across lesion types (DAF vs. DDAF), measurement methods (automated vs. manual), and data scales (raw vs. transformed) to understand the contribution of each factor to reliability and concordance. Data Transformation and Rationale All DAF and DDAF measurements were initially inspected for normality and skewness. Histograms, Shapiro–Wilk tests, and calculations of skewness indicated that the data were substantially right-skewed, with a small number of markedly larger lesions inflating the mean and variance. Such heavy skew can undermine parametric assumptions in reliability and agreement analyses. Accordingly, a square-root transformation (√x) was applied to all raw measurements (DAF, DDAF) to stabilize variance and reduce outlier influence. Nonetheless, both raw and transformed analyses are presented in this study to allow the clinical community to interpret the lesion measurements using the original scale. Specifically, we report inter- and intra-grader reliability and method-comparison results (e.g., ICC, Bland–Altman limits, Concordance Correlation Coefficients) on the untransformed (raw) data and on the square-root transformed data. This approach ensures that readers can (1) gauge performance under standard clinical units and (2) appreciate any improvements in statistical robustness afforded by the transformation. Results FAF images from a total of 316 eyes of 158 subjects were assessed in the study, with a mean age of 15.4 ± 2.25 years (range, 12–21 years). The cohort included 91 males (57.6%) and 67 females (42.4%). Most of participants identified as Asian (58.9%), followed by Caucasian (34.2%), with a small proportion identifying as Other (5.1%) or declining to report race (1.9%) ( Table 1 ). View this table: View inline View popup Download powerpoint Table 1. Baseline demographic and clinical characteristics of study participants. Values are reported as mean ± standard deviation for continuous variables and as counts with percentages for categorical variables. BCVA = best-corrected visual acuity. Images from two eyes (2 subjects) were deemed ungradable by all graders due to insufficient quality to perform any type of analysis. Additional 12 images were deemed ungradable for manual grading by at least two of the three graders in the manual graders group. Table 2 summarizes the computed values for each grader and endpoint, both on the raw scale (DAF and DDAF) and after square root transformation (sqDAF and sqDDAF). Figure 1 shows representative examples of manual and automated delineation of DAF and DDAF lesions across a range of lesion sizes and configurations. Download figure Open in new tab Figure 1. Comparison of manual and automated delineation of decreased autofluorescence (DAF) and definitely decreased autofluorescence (DDAF) lesions on fundus autofluorescence (FAF) imaging in four representative eyes with Stargardt disease. Each row (Panels A–D, top to bottom) represents a different subject. For each, FAF images show lesion segmentation by two representative manual graders (Grader 1, left; Grader 2, center) and one representative automated grader (right). In manual delineations, DDAF regions are outlined in pink and DAF regions are outlined in orange. In automated segmentations, DDAF is defined using a >90% darkness threshold relative to the optic disc (fuchsia), and DAF is defined using a >75% threshold (dark blue). Additional contours reflect alternative thresholds used for visualization only: >70% (light blue) and >95% (red), not included in analysis. Green concentric circles represent ETDRS-like reference zones. Panel A : Manual graders show close agreement for both DAF and DDAF, with minor differences in boundary irregularity. The automated delineation mirrors both graders, with slightly larger nasal DDAF extension. Panel B : Manual graders agree on overall DAF extent, with Grader 1 identifying a slightly larger DDAF area. The automated segmentation closely matches Grader 1 for both DAF and DDAF. Panel C : Both manual graders agree on the presence and extent of DAF, but only Grader 1 identifies a DDAF lesion. The automated segmentation aligns with both graders for DAF and with Grader 1 regarding the presence and boundaries of DDAF. Panel D : A large central DAF lesion is consistently delineated by both manual graders, though there is marked disagreement regarding the extent of DDAF. The automated segmentation corresponds well with both graders for DAF and identifies DDAF boundaries that fall between those defined by the manual graders. These examples highlight the variability in manual grading, particularly for DDAF boundaries, and the potential of automated methods to provide reproducible and standardized lesion quantification. Additional threshold contours may offer future utility for stratified staging or progression modeling. View this table: View inline View popup Download powerpoint Table 2. Descriptive statistics for both Decreased Autofluorescence (DAF) and Definitely Decreased Autofluorescence (DDAF) measurements across five graders (two for automated and three for Manual). For each endpoint, both raw lesion areas (in mm²) and their square-root transformed values (sqDAF and sqDDAF) are reported to support statistical analysis and comparability across methods. Intra-Grader Repeatability Automated measurements demonstrated excellent intra-grader repeatability for both DAF and DDAF across graders G1 and G2, with ICCs approaching 0.99 in both raw and transformed data. For example, G1 achieved ICCs of 0.988 (DAF) and 0.989 (DDAF), with corresponding SEM values of 0.69 mm² and 0.44 mm², and MDC/CR values of 1.90 mm² and 1.23 mm², respectively. These metrics remained stable after square-root transformation; the observed reductions in SEM and MDC/CR primarily reflect the compressed measurement scale rather than an intrinsic improvement in repeatability. G2 showed nearly identical results, reinforcing the reproducibility of the automated approach across both lesion types and indicating minimal operator-related variability. Manual graders also achieved generally high repeatability, particularly for DAF, though variability was greater than in the automated method and differed by lesion type and grader. For DAF, ICCs ranged from 0.909 (G1) to 0.974 (G3) in raw data. While transformation consistently reduced the absolute SEM and MDC/CR values due to scale compression, the degree to which it improved relative measurement precision varied across graders. G1 showed the highest SEM and MDC/CR in raw DAF (2.41 and 6.69 mm²), with notable reduction after transformation (0.28 and 0.78 mm²), though still higher than other graders. G2 and G3 exhibited similar improvements post-transformation, with already stronger baseline performance. Manual DDAF measurements were less consistent. G1’s ICC improved from 0.837 (raw) to 0.940 (transformed), reflecting heteroscedasticity in lesion size distribution, with corresponding reductions in SEM and MDC/CR. G2 maintained high reliability for DDAF in both scales (ICC 0.947 raw; 0.948 transformed), while G3 showed weaker repeatability, with ICC decreasing from 0.842 to 0.816 after transformation. In G3’s case, the transformation did not confer benefit and may have exposed underlying inconsistencies, as SEM and MDC/CR remained relatively high (1.77 and 4.91 mm² raw; 0.49 and 1.35 mm² transformed). Overall, intra-grader ICCs ranged from 0.837 to 0.990 in raw data and improved further after transformation (0.816–0.993). The transformation generally reduced measurement error and improved reliability across most graders and endpoints, particularly for manual measurements, although its effect was not uniformly beneficial. Complete metrics are provided in Table 3 . View this table: View inline View popup Download powerpoint Table 3. Intra-grader repeatability metrics for Decreased Autofluorescence (DAF) and Definitely Decreased Autofluorescence (DDAF) measurements using raw and square-root-transformed data. The table summarizes intra-grader reliability for three graders (G1–G3) using either manual or automated methods. For each grader–endpoint combination, the table presents the intraclass correlation coefficient (ICC), pooled standard deviation (SD), standard error of measurement (SEM), and minimum detectable change or coefficient of repeatability (MDC/CR), calculated separately for raw and square-root-transformed scales. Higher ICC values and lower SEM/MDC values indicate greater repeatability. Square-root transformation consistently reduced SEM and MDC across both endpoints, particularly for manual measurements of larger lesions, suggesting improved precision and stability of repeated measurements following transformation. These findings were further supported by Bland–Altman plots, which visually confirmed the consistency of repeated measurements within each grader (Supplemental Figure 2 through Supplemental Figure 5 ). For the automated method, both graders demonstrated minimal bias and narrow limits of agreement in raw and square-root-transformed scales for both DAF and DDAF (Supplemental Figure 2 and Supplemental Figure 3 ). The spread of differences was small and evenly distributed around the mean, with little evidence of heteroscedasticity. In contrast, manual graders exhibited greater variability in repeated measurements (Supplemental Figure 4 and Supplemental Figure 5 ), especially in raw DAF and DDAF, where wider limits of agreement and skewed distributions were apparent, most notably in Grader 1’s raw DAF and Grader 3’s raw DDAF plots. Square-root transformation improved the agreement patterns for most manual graders by reducing outlier influence and narrowing the limits of agreement, though some grader-specific inconsistencies persisted. Together, these plots reinforce the numerical reliability metrics and highlight the enhanced reproducibility of the automated approach, as well as the benefit, but not guarantee, of variance stabilization via transformation in manual assessments. Download figure Open in new tab Figure 2. Intra-grader repeatability for Decreased Autofluorescence (DAF) – Automated graders. Bland–Altman plots for raw (top row) and square-root-transformed (bottom row) DAF measurements from two automated graders (left: Grader 1, right: Grader 2). The plots show minimal bias, tight limits of agreement (LoA), and no clear trend of heteroscedasticity, indicating excellent repeatability of the automated method. Download figure Open in new tab Figure 3. Intra-grader repeatability for Definitely Decreased Autofluorescence (DDAF) – Automated graders. Bland–Altman plots for raw (top) and square-root-transformed (bottom) DDAF measurements for automated Grader 1 (left) and Grader 2 (right). Both sets demonstrate low bias and narrow LoA, with improved uniformity in the transformed plots, reflecting high internal consistency of the automated algorithm. Download figure Open in new tab Figure 4. Intra-grader repeatability for Decreased Autofluorescence (DAF) – Manual graders. Bland– Altman plots for raw (top row) and square-root-transformed (bottom row) DAF measurements from three manual graders (left to right: Graders 1–3). While transformation improves uniformity and narrows LoA, raw plots show greater dispersion, especially for larger lesion sizes, indicating variable repeatability across graders. Download figure Open in new tab Figure 5. Intra-grader repeatability for Definitely Decreased Autofluorescence (DDAF) – Manual graders. Bland–Altman plots for raw (top) and square-root-transformed (bottom) DDAF from manual Graders 1–3 (left to right). Manual measurements exhibit wider LoA and occasional outliers in raw data; square-root transformation improves agreement for Graders 1 and 2, while variability remains notable for Grader 3. Within-Method Reliability (Inter-Grader Repeatability) Inter-grader repeatability metrics are summarized in Table 4 , including intraclass correlation coefficients (ICC) for both single- and average-measure reliability (two-way random effects, absolute agreement), along with pooled standard deviation (SD), standard error of measurement (SEM), and minimal detectable change (MDC) at the 95% confidence level. View this table: View inline View popup Download powerpoint Table 4. Intraclass correlation coefficients (ICCs) for pairwise comparisons of graders on both Decreased Autofluorescence (DAF) and Definitely Decreased Autofluorescence (DDAF), using both raw and square-root transformed measurements. For each comparison, single-measure and average-measure ICC estimates are reported, along with corresponding Standard Error of Measurement (SEM) and Minimal Detectable Change (MDC) at 95% confidence. Values are derived from a two-way random-effects model assuming absolute agreement, reflecting the reliability of individual ratings and the mean of two ratings. For DAF, agreement between the two automated graders (Auto G1 vs. G2) was excellent, with ICCs of 0.989 (raw) and 0.988 (square-root), and average-measure ICCs of 0.994 for both scales ( Table 4 ). SEM and MDC values were low (≤ 0.72 mm² and ≤ 1.99 mm², respectively), and further decreased following square-root transformation, indicating minimal measurement error. Manual grader pairs demonstrated moderate-to-high repeatability, with single-measure ICCs ranging from 0.764 to 0.879 in the raw scale and 0.899 to 0.909 in the square-root scale. Average-measure ICCs improved accordingly (ranging from 0.866 to 0.936 raw and 0.947–0.952 transformed). Square-root transformation reduced the SEM and MDC values across all manual comparisons due in part to the rescaling effect; however, when paired with improvements in ICCs and Bland–Altman agreement, this suggests a potential enhancement in relative measurement consistency. Bland–Altman plots (Supplemental Figure 6 ) support these findings. For raw DAF, the automated graders showed near-zero mean difference and narrow limits of agreement (LoA), with few outliers or signs of heteroscedasticity. Manual comparisons displayed broader LoA and greater dispersion at higher lesion sizes. In the square-root-transformed plots, differences appeared more uniform across the range of values. The automated pair retained minimal bias and tight LoA, while manual pairs showed reduced spread and fewer extreme outliers, indicating that transformation helped stabilize measurement variance. Download figure Open in new tab Figure 6. Bland–Altman plots of raw Decreased Autofluorescence (DAF) measurements (top row) and square-root DAF (bottom row), comparing two automated graders (left-most plots) and each pair of manual graders (middle and right plots). The central dashed line indicates the mean difference (bias), while the upper and lower dashed lines represent the 95% limits of agreement (LoA). For the manual comparisons, LoA are visibly wider in the raw scale, particularly at larger average lesion areas, but become narrower following the square-root transformation. The automated graders consistently show minimal bias and narrow LoA in both raw and transformed scales. Similar trends were observed for DDAF. Automated graders achieved ICCs of 0.988 (raw) and 0.992 (square-root), with average-measure ICCs of 0.994 and 0.996, respectively. SEM and MDC values remained consistently low (e.g., SEM ≤ 0.54 mm²; MDC ≤ 1.48 mm²), reaffirming the method’s precision ( Table 4 ). Manual pairs showed high single-measure ICCs (0.892–0.939 raw) and modestly lower values in the transformed scale (0.867–0.922), with average-measure ICCs remaining strong (0.943–0.968 raw; 0.929–0.959 transformed). As with DAF, square-root transformation reduced SEM and MDC values for most manual comparisons due to the scale reduction; in conjunction with improved agreement metrics, this supports its utility for mitigating heteroscedasticity. Bland–Altman plots for DDAF (Supplemental Figure 7 ) mirrored the DAF pattern. Automated graders exhibited minimal bias and tight LoA in both raw and transformed scales. Manual graders displayed broader LoA in raw data, particularly at larger lesion sizes, but showed visibly reduced dispersion after transformation. These results reinforce that square-root transformation mitigates heteroscedasticity and improves agreement, especially for manual assessments. Download figure Open in new tab Figure 7. Bland–Altman plots of Definitely Decreased Autofluorescence (DDAF) measurements, showing (top row) raw DDAF for the automated graders (left) and each pair of manual graders (center and right), followed by (bottom row) square-root DDAF. The central dashed lines represent the mean differences (bias), while the upper and lower dashed lines show the 95% limits of agreement (LoA). Automated graders exhibit negligible bias and narrow LoA in both raw and transformed scales, whereas manual graders display greater variability in the raw data, especially at higher lesion sizes, but benefit from noticeably reduced scatter and tighter LoA after square-root transformation. Agreement Analysis (Lin’s CCC and Passing–Bablok) Agreement between the automated and manual methods was assessed using eyes that were gradable by both approaches and had lesions fully contained within the central 6 mm region of the macula. This constraint is intrinsic to the automated algorithm, which is designed to restrict lesion detection to the central 6 mm area. Inclusion of eyes with lesions extending beyond this region would have introduced a ceiling effect in the automated measurements, potentially distorting the concordance metrics. To ensure valid comparisons, only eyes with lesions measurable by both methods within the same anatomical boundary were included in the analysis. As a result, 14 eyes were excluded, and the final agreement analysis was conducted on data from 288 eyes. Lin’s Concordance Coefficient (CCC) indicated strong agreement for both DAF and DDAF (Supplemental Figure 8 ). In the raw scale, CCC values were approximately 0.91 for DAF and 0.96 for DDAF ( Table 5 ), reflecting high concordance between the new automated approach and the manual reference. After square-root transformation, CCC values remained high (0.92 for sqDAF and 0.94 for sqDDAF), confirming that the methods retained strong alignment even when skewness was reduced. Overall, these results reinforce that the automated algorithm’s measurements are largely consistent with manual grading across both lesion endpoints. Download figure Open in new tab Figure 8. Scatter plots comparing automated and manual mean measurements for Decreased Autofluorescence (DAF) and Definitely Decreased Autofluorescence (DDAF) in both raw and square-root-transformed scales. Each panel displays individual data points (AutoMean vs. ManuMean) for a specific endpoint: raw DAF (top left), raw DDAF (top right), square-root DAF (bottom left), and square-root DDAF (bottom right). The dotted orange line represents the line of identity (Y = X) used in Lin’s Concordance Correlation analysis, while the dotted black line represents the Passing–Bablok regression line fitted to the same data. Visual agreement is strongest where both lines are closely aligned and data points cluster tightly along them. The modest divergence of the Passing–Bablok line from the identity line in some panels reflects slight proportional or constant bias, particularly in raw DAF and sqDAF. View this table: View inline View popup Download powerpoint Table 5. Summary of agreement analyses between Automated (AutoMean) and Manual (ManuMean) graders for Decreased Autofluorescence (DAF) and Definitely Decreased Autofluorescence (DDAF) (raw and square-root scales). Lin’s CCC quantifies the overall concordance, while Passing–Bablok regression evaluates potential constant or proportional bias between the two methods. 1. Lin’s CCC is computed using the formula ((2ρσ_x σ_y)/(σ_x^2+σ_y^2+(μ_x-μ_y)^2)), where(ρ) is Pearson’s correlation, (σ_x,σ_y) are SDs, and (μ_x,μ_y) are means. 2. Passing–Bablok Slope: Values close to 1 suggest minimal proportional bias; values significantly under or above 1 imply under-/overestimation by one method relative to the other. 3. Passing–Bablok Intercept: Values close to 0 suggest minimal constant bias; positive or negative intercepts point to a fixed offset between the two methods. Passing–Bablok regression (Supplemental Figure 8 ) revealed modest intercepts and slopes close to 1 across all comparisons, indicating limited systematic bias. For raw DAF, the slope was 0.945 with an intercept of 0.665, while for sqDAF it was 0.880 with a lower intercept of 0.369 ( Table 5 ), indicating minor proportional and constant deviations, respectively. Similar trends were observed for DDAF (Supplemental Figure 8 ): the raw scale showed a slope of 0.910 and an intercept of 0.230, while the square-root scale yielded a slope of 0.864 and an intercept of 0.247 ( Table 5 ). These parameters suggest slightly stronger alignment in the raw data but no indication of clinically meaningful deviation in either scale. Discussion This study introduces and validates a novel automated method for quantifying DAF and DDAF in Stargardt disease, demonstrating highly reproducible measurements with minimal grader dependency. ICCs near 0.99 and consistently low SEM and MDC values across graders and lesion types underscore the novel method’s robustness. These results support the use of automated methods in both clinical and research settings where consistency, speed, and scalability are paramount. In contrast, manual measurements, though generally reliable for DAF, exhibited greater intra- and inter-grader variability, especially for DDAF. This variability likely stems from the inherent difficulty of delineating lesion borders manually, particularly for smaller or irregular DDAF lesions. The variability was most pronounced in Grader 3’s assessments of DDAF, where transformation failed to substantially improve reliability, possibly reflecting systematic grading inconsistencies. Square-root transformation substantially improved repeatability for most manual measurements by reducing heteroscedasticity. This effect was consistent across graders and endpoints, lowering SEM and MDC and raising ICCs, although not universally. Although square-root transformation reduced SEM and MDC values, these reductions must be interpreted relative to the transformed scale. Improvements in ICCs and agreement plots offer stronger evidence of genuine gains in measurement precision. The transformation had less benefit when systematic bias or grader inconsistency was present, as seen in Grader 3’s manual DDAF data. These findings suggest that transformation is useful for mitigating random variability, but may not fully correct for systematic subjectivity. The difference in reproducibility between DAF and DDAF further highlights the challenges of lesion delineation. DAF measurements were generally more consistent, likely because they are larger, better defined due to better contrast between the lesion boundaries and the normal retinal background, and less susceptible to interpretation bias. In contrast, DDAF lesions in this young cohort were often smaller, patchy, and less well demarcated against a background of QDAF, where the distinction between areas with gray levels >90% and <90% of the optic disc reference is less obvious, contributing to lower agreement. This contrasts with findings from the ProgStar study, which reported higher reproducibility for DDAF, possibly due to inclusion of older patients with more advanced, consolidated lesions. The younger DRAGON cohort, by design, represented earlier disease stages, in which DDAF lesion margins may not yet be clearly defined, increasing manual grading difficulty. Despite these challenges, the square-root transformation consistently enhanced the reliability of manual assessments, particularly for DDAF. This underscores its value in reducing the impact of outliers and scale-related variance, improving the utility of manual grading in settings where automation is not available. Agreement between the automated and manual methods was strong across both lesion types, with Lin’s CCC values exceeding 0.90 in all comparisons. Although Passing–Bablok regression revealed mild proportional and constant bias, the deviations were not clinically significant. The automated method’s use of fixed thresholds (e.g., ≥75% of optic disc darkness) may have contributed to slightly elevated lesion size estimates, particularly in DAF. Nonetheless, these small differences are likely correctable through minor calibration if tighter alignment with manual reference standards is desired. In addition to its measurement precision, the automated method demonstrated greater robustness. Only two images (0.6%) were ungradable by the automated method, compared to 14 (4.4%) by manual grading. The automated approach also yielded fewer outlier exclusions, supporting its capacity to handle image variability and enhance data inclusion in large-scale studies. Taken together, these findings have important implications. Automated methods offer a consistent and scalable solution for lesion quantification and may significantly reduce intergrader variability in clinical trials. For centers relying on manual grading, applying square-root transformation can improve reliability, particularly for smaller or less defined lesions. However, manual assessment of DDAF remains more variable, underscoring the importance of adopting automated approaches when available. Limitations This study has several limitations. First, its retrospective design inherently restricted the study cohort to the demographic and clinical characteristics defined by the original trial’s inclusion and exclusion criteria, such as age at onset, projected visual acuity, and estimated lesion size. As a result, the findings may not be fully generalizable to the broader Stargardt disease population, particularly those with earlier-stage or more advanced phenotypes. Second, the cross-sectional nature of the study precluded assessment of the longitudinal stability or tracking performance of the automated method compared to the manual approach. Finally, although the automated method demonstrated strong performance, it was evaluated within the context of standardized imaging protocols and a controlled trial environment; its applicability in real-world clinical settings with greater variability in image quality and acquisition remains to be established. Data Availability All data produced in the present study are available upon reasonable request to the authors. Footnotes Meeting Presentation The abstract of this research was presented at the 2024 ARVO Annual Meeting, held in Seattle, WA, May 5-9, 2024. Financial Support None received. Conflict of Interest The authors have disclosed their conflict of interest as per ICMJE criteria. Precis An automated method for measuring areas of decreased autofluorescence in Stargardt disease demonstrates high agreement with manual grading and offers a more consistent, efficient approach for lesion quantification in clinical studies. References 1. ↵ Tanna P , Strauss RW , Fujinami K , Michaelides M . Stargardt disease: clinical features, molecular genetics, animal models and therapeutic options . Br J Ophthalmol . 2017 ; 101 ( 1 ): 25 – 30 . OpenUrl Abstract / FREE Full Text 2. ↵ Strauss RW , Ho A , Munoz B , Cideciyan AV , Sahel JA , Sunness JS , et al. The Natural History of the Progression of Atrophy Secondary to Stargardt Disease (ProgStar) Studies: Design and Baseline Characteristics: ProgStar Report No. 1 . Ophthalmology . 2016 ; 123 ( 4 ): 817 – 28 . OpenUrl CrossRef PubMed 3. ↵ Fujinami K , Lois N , Mukherjee R , McBain VA , Tsunoda K , Tsubota K , et al. A longitudinal study of Stargardt disease: quantitative assessment of fundus autofluorescence, progression, and genotype correlations . Invest Ophthalmol Vis Sci . 2013 ; 54 ( 13 ): 8181 – 90 . OpenUrl Abstract / FREE Full Text 4. ↵ Georgiou M , Kane T , Tanna P , Bouzia Z , Singh N , Kalitzeos A , et al. Prospective Cohort Study of Childhood-Onset Stargardt Disease: Fundus Autofluorescence Imaging, Progression, Comparison with Adult-Onset Disease, and Disease Symmetry . Am J Ophthalmol . 2020 ; 211 : 159 – 75 . OpenUrl CrossRef PubMed 5. ↵ Strauss RW , Kong X , Ho A , Jha A , West S , Ip M , et al. Progression of Stargardt Disease as Determined by Fundus Autofluorescence Over a 12-Month Period: ProgStar Report No. 11 . JAMA ophthalmology . 2019 ; 137 ( 10 ): 1134 – 45 . OpenUrl PubMed 6. ↵ Jauregui R , Nuzbrokh Y , Su PY , Zernant J , Allikmets R , Tsang SH , et al. Retinal Pigment Epithelium Atrophy in Recessive Stargardt Disease as Measured by Short-Wavelength and Near-Infrared Autofluorescence . Translational vision science & technology . 2021 ; 10 ( 1 ): 3 . OpenUrl 7. ↵ Strauss RW , Ho A , Jha A , Fujinami K , Michaelides M , Cideciyan AV , et al. Progression of Stargardt Disease as Determined by Fundus Autofluorescence Over a 24-Month Period (ProgStar Report No. 17) . Am J Ophthalmol . 2023 ; 250 : 157 – 70 . OpenUrl CrossRef PubMed 8. ↵ Schmitz-Valckenberg S , Brinkmann CK , Alten F , Herrmann P , Stratmann NK , Gobel AP , et al. Semiautomated image processing method for identification and quantification of geographic atrophy in age-related macular degeneration . Invest Ophthalmol Vis Sci . 2011 ; 52 ( 10 ): 7640 – 6 . OpenUrl Abstract / FREE Full Text 9. ↵ Zhao PY , Branham K , Schlegel D , Fahim AT , Jayasundera KT . Automated Segmentation of Autofluorescence Lesions in Stargardt Disease . Ophthalmol Retina . 2022 ; 6 ( 11 ): 1098 – 104 . OpenUrl CrossRef PubMed 10. ↵ Feuer WJ , Yehoshua Z , Gregori G , Penha FM , Chew EY , Ferris FL , et al. Square root transformation of geographic atrophy area measurements to eliminate dependence of growth rates on baseline lesion measurements: a reanalysis of age-related eye disease study report no. 26 . JAMA ophthalmology . 2013 ; 131 ( 1 ): 110 – 1 . OpenUrl PubMed 11. ↵ Mones J , Biarnes M . The Rate of Progression of Geographic Atrophy Decreases With Increasing Baseline Lesion Size Even After the Square Root Transformation . Translational vision science & technology . 2018 ; 7 ( 6 ): 40 . OpenUrl PubMed View the discussion thread. Back to top Previous Next Posted July 07, 2025. Download PDF Data/Code Email Thank you for your interest in spreading the word about medRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following Automated Quantification of Decreased FAF in Stargardt Disease: Validation of a Novel Method Compared to Manual Grading Standards Message Subject (Your Name) has forwarded a page to you from medRxiv Message Body (Your Name) thought you would like to see this page from the medRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share Automated Quantification of Decreased FAF in Stargardt Disease: Validation of a Novel Method Compared to Manual Grading Standards Mohamed I. Ahmed , Hikmet Yucel , Rubbia Afridi , Thales A.C. de Guimarães , Isabel Sendino-Tenorio , Nam V. Nguyen , Kiran Romaisa , Sidrah Khan , Ufaq Khan , Syeda Sharaf un Nisa , Mauro Campigotto , Amir Hariri , Michel Michaelides , Hendrik P.N. Scholl , Nathan Mata , Quan D. Nguyen , Yasir J. Sepah medRxiv 2025.07.07.25330927; doi: https://doi.org/10.1101/2025.07.07.25330927 Share This Article: Copy Citation Tools Automated Quantification of Decreased FAF in Stargardt Disease: Validation of a Novel Method Compared to Manual Grading Standards Mohamed I. Ahmed , Hikmet Yucel , Rubbia Afridi , Thales A.C. de Guimarães , Isabel Sendino-Tenorio , Nam V. Nguyen , Kiran Romaisa , Sidrah Khan , Ufaq Khan , Syeda Sharaf un Nisa , Mauro Campigotto , Amir Hariri , Michel Michaelides , Hendrik P.N. Scholl , Nathan Mata , Quan D. Nguyen , Yasir J. Sepah medRxiv 2025.07.07.25330927; doi: https://doi.org/10.1101/2025.07.07.25330927 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Ophthalmology Subject Areas All Articles Addiction Medicine (569) Allergy and Immunology (863) Anesthesia (300) Cardiovascular Medicine (4441) Dentistry and Oral Medicine (444) Dermatology (383) Emergency Medicine (609) Endocrinology (including Diabetes Mellitus and Metabolic Disease) (1510) Epidemiology (15229) Forensic Medicine (30) Gastroenterology (1126) Genetic and Genomic Medicine (6607) Geriatric Medicine (668) Health Economics (998) Health Informatics (4541) Health Policy (1369) Health Systems and Quality Improvement (1613) Hematology (543) HIV/AIDS (1266) Infectious Diseases (except HIV/AIDS) (15923) Intensive Care and Critical Care Medicine (1103) Medical Education (623) Medical Ethics (147) Nephrology (668) Neurology (6605) Nursing (346) Nutrition (998) Obstetrics and Gynecology (1146) Occupational and Environmental Health (957) Oncology (3334) Ophthalmology (974) Orthopedics (369) Otolaryngology (420) Pain Medicine (436) Palliative Medicine (130) Pathology (663) Pediatrics (1693) Pharmacology and Therapeutics (692) Primary Care Research (711) Psychiatry and Clinical Psychology (5448) Public and Global Health (9236) Radiology and Imaging (2200) Rehabilitation Medicine and Physical Therapy (1370) Respiratory Medicine (1196) Rheumatology (594) Sexual and Reproductive Health (713) Sports Medicine (530) Surgery (712) Toxicology (99) Transplantation (289) Urology (265) (function(){function c(){var b=a.contentDocument||a.contentWindow.document;if(b){var d=b.createElement('script');d.innerHTML="window.__CF$cv$params={r:'a017e0614b23df94',t:'MTc3OTc0NzA5Mg=='};var a=document.createElement('script');a.src='/cdn-cgi/challenge-platform/scripts/jsd/main.js';document.getElementsByTagName('head')[0].appendChild(a);";b.getElementsByTagName('head')[0].appendChild(d)}}if(document.body){var a=document.createElement('iframe');a.height=1;a.width=1;a.style.position='absolute';a.style.top=0;a.style.left=0;a.style.border='none';a.style.visibility='hidden';document.body.appendChild(a);if('loading'!==document.readyState)c();else if(window.addEventListener)document.addEventListener('DOMContentLoaded',c);else{var e=document.onreadystatechange||function(){};document.onreadystatechange=function(b){e(b);'loading'!==document.readyState&&(document.onreadystatechange=e,c())}}}})();
Text is read by the "Ask this paper" AI Q&A widget below.
Extraction quality varies by source — PMC NXML preserves structure
cleanly, OA-HTML may include some navigation residue, and OA-PDF can
have broken hyphenation. The publisher copy
(via DOI)
is the canonical version.