TSGMA: identification of macro associations from global data to build global MA networks

preprint OA: closed
📄 Open PDF Full text JSON View at publisher

Abstract

The growing availability of globally harmonized datasets offers unprecedented opportunities to identify population-level risk factors, yet systematic tools for macro-scale association analysis remain scarce. Here, I introduce the concept of “macro association” (MA) and propose a three-tiered framework: Three Stars Global Macro Association-analysis (TSGMA), which integrates correlation, partial correlation adjusted for confounders, and temporal lag analysis to rapidly and robustly rank global associations. Applying TSGMA to dietary and health data from 159 countries or regions, I identify strong links between diet and three cardiometabolic markers. Animal fats group, red meat, and eggs show robust, time-lagged associations with elevated non-HDL cholesterol; sugar, cereals, and poultry meat are associated with increased diabetes prevalence; while starchy roots and pulses consistently exhibit protective associations. TSGMA not only confirms established patterns but also reveals overlooked signals, offering a scalable approach to construct integrative global “MA networks” and enabling hypothesis generation from open-access “macro data”. Graphical Abstract
Full text 73,407 characters · extracted from preprint-html · click to expand
TSGMA: identification of macro associations from global data to build global MA networks | medRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-P4HH5NV'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search TSGMA: identification of macro associations from global data to build global MA networks View ORCID Profile Hongyue Ma doi: https://doi.org/10.1101/2025.06.09.25329257 Hongyue Ma 1 Haide College, Ocean University of China , Qingdao, China 2 School of Life Sciences, Westlake University , Hangzhou, Zhejiang, China Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Hongyue Ma For correspondence: mahongyue_edu{at}126.com Abstract Full Text Info/History Metrics Supplementary material Data/Code Preview PDF Abstract The growing availability of globally harmonized datasets offers unprecedented opportunities to identify population-level risk factors, yet systematic tools for macro-scale association analysis remain scarce. Here, I introduce the concept of “macro association” (MA) and propose a three-tiered framework: Three Stars Global Macro Association-analysis (TSGMA), which integrates correlation, partial correlation adjusted for confounders, and temporal lag analysis to rapidly and robustly rank global associations. Applying TSGMA to dietary and health data from 159 countries or regions, I identify strong links between diet and three cardiometabolic markers. Animal fats group, red meat, and eggs show robust, time-lagged associations with elevated non-HDL cholesterol; sugar, cereals, and poultry meat are associated with increased diabetes prevalence; while starchy roots and pulses consistently exhibit protective associations. TSGMA not only confirms established patterns but also reveals overlooked signals, offering a scalable approach to construct integrative global “MA networks” and enabling hypothesis generation from open-access “macro data”. Download figure Open in new tab 1. Introduction Understanding the myriad factors that influence human health is a central challenge in science 1 . Lifestyle and environmental factors such as diet, medications, physical activity, sleep, air quality, and work conditions interact in complex ways to affect health outcomes 1 – 3 . Broadly mapping these relationships can greatly improve our ability to prevent disease and promote wellness. Identifying potential risk factors and protective factors through comprehensive association analysis can guide more focused research and interventions. Historically, early epidemiology relied on simple ecological correlations or rule-of-thumb observations 4 . The most famous classic example is the discovery of geographic consistency between smoking prevalence and lung cancer mortality in the middle ages in the twentieth century, which proved the heuristic value of crude analyses 5 . But since then, limitations resulting from issues such as sparse data, rudimentary statistics, and profound confounding restricted both scope and credibility have come to light 6 – 8 . Over the following decades, controlled trials, large prospective cohorts, and meta-analyses transformed risk assessment into a rigorous discipline 9 – 11 . Landmark resources such as the UK Biobank and the Global Burden of Disease study now permit precise causal inferences to be made about prespecified hypotheses 1 , 10 . However, this depth comes at the cost of breadth: cohorts are expensive to assemble, most cover a narrow range of populations, and by design they can only test a limited group of candidate exposures 10 , 12 – 14 . As a result, modern studies are better at confirming known risks but less adept at identifying new ones. Meanwhile, digitization has generated an unprecedented number of open national statistical databases, such as Our World in Data ( https://ourworldindata.org ), the World Bank ( https://data.worldbank.org ), FAO ( https://www.fao.org/faostat/en/#data/FBS ) and International Agency for Research on Cancer ( https://www.iarc.who.int ), as well as unprecedented conditions for data collection. These “macro data” (MD)are numerically simple, globally harmonized and directly interpretable by both scientists and the public, however, they are still lack being systematically mined 15 – 17 . Traditional fine-grained epidemiological methods are not applicable to the analysis of these data, and exploratory, simple correlation analyses have been gradually abandoned as yielding results that are inferior to those of precise epidemiological studies 6 , 8 , 15 , 17 – 19 . Broad exploratory studies of cross-cutting risk factors, once central to public health research, are increasingly seen as too “ecological” or methodologically difficult to navigate 19 – 21 . Unfortunately, it is in the very era that provides the richest data that comprehensive, integrated broad explorations of risk factors are retreating. Thus, against the backdrop of data richness that exceeds any previous era, I propose a novel concept: “macro association” (MA), which I define as a wide range of repeatable associations detectable on a domain-wide scale in the context of MD. Moreover, I name the systematic analytical mining of MA as “macro association analysis” (MA-analysis), hence, MA-analysis aims to uncover such associations in large publicly available datasets as well as in various accessible data sources, and ultimately to build a comprehensive domain-wide “macro association network” (MA network) across various domains, such as health, economy, environment, climate, and other relevant domains. Among the numerous possible dimensions of MA, the country-based global dimension is currently the most attractive due to the increasing completeness of the data, the simplicity of the figures, and the direct relevance to human policies. However, simple correlations are easily distorted by confounding and reverse causation, and gold standard epidemiological methods are not applicable to the analysis of this type of data, making MA-analysis in these data particularly difficult 6 , 22 . To bridge this gap, I developed the Three Stars Global Macro Association-analysis (TSGMA), a three-step screening methodology that grades pairs of associations between any global variables from zero to three stars by sequentially applying crude correlations (i, one star), confounder-controlled correlations (ii, two stars), and time lag analyses (iii, three stars). TSGMA provides an accurate, fast, and efficient way to perform MA-analysis of global dimensional data on a country basis, and lays the foundation for the construction of global macro association networks. As a proof of concept, I applied TSGMA to the analysis of associations between dietary components and three cardiometabolic markers (hypertension, diabetes, and non-HDL-C levels) in 159 countries or regions (1992-2017). The methodology recovered typical associations, such as those between sugar and diabetes as well as alcohol and hypertension. At the same time, it revealed some overlooked patterns, such as a population lag of about 15 years between animal fat intake and non-HDL-C levels. These results illustrate how TSGMA can translate global MD into actionable hypotheses and lay the groundwork for building comprehensive, cross-cutting global MA networks. In addition, I propose a structured framework for the construction of a global MA network. The process begins with applying TSGMA to individual databases to build internal MA networks, which are then progressively extended across multiple repositories. These independent networks are subsequently harmonized and iteratively integrated using artificial intelligence and advanced analytics into a unified global MA network. This framework transforms decentralized, open-access datasets into a cohesive, dynamic system capable of uncovering the intertwined drivers of planetary health, development, and sustainability. 2. Methods 2.1 Overview of the TSGMA approach TSGMA is a three steps analytical methodology designed to explore and categorize associations between variables accurately and quickly with global data such as multiple years of country or region level data. Figure 1 provides a schematic overview of the TSGMA workflow. Briefly, the steps are as follows: (1) One star analysis (correlation analysis): Correlation coefficients between variables are calculated to screen for significantly correlated variables ( p < 0.05). (2) Two stars analysis (partial correlation analysis): Key confounding variables are adjusted using partial correlation analysis to assess whether the observed associations persist after controlling for confounding effects ( p < 0.05). (3) Three stars analysis (temporal correlation analysis): Temporal patterns of paired variables are examined to determine whether the change in one of the variables is consistently similar to the trend of the other variable with a relatively stable time lag. At each step, candidate variables are screened and given a “score” to indicate the strength of evidence for a true association: “zero star” = no association, “one star” = suggestive association, “two stars” = robust association after confounder adjustment, and “three stars” = probable causality with temporal priority. Download figure Open in new tab Figure 1. Overview of the TSGMA workflow. (i) Define the scope of the analysis and identify all variables of interest. (ii) Collect data for each variable from all countries or regions for which data are available. Data are cleaned and screened for countries or regions whose data are as complete as possible for a given period, and the data for each variable are then averaged. (iii) Calculate the correlation coefficient between each candidate variable. Pairs of variables showing significant correlation ( p < 0.05) are screened for two stars analysis and uncorrelated variables (zero star) are discarded. (iv) Identify key potential confounders or key confounder representations based on the properties and meaning of the variables (for global variables, a classic example is socio-economic status, where GDP per capita is a useful key representation indicator 23 ). Partial correlation coefficients between one star associated variables are calculated to control for confounders. Variables that maintain significant associations ( p < 0.05) are upgraded to two stars associations, while those that lose significance are considered only at the one star association level. (v) A time-lag analysis is conducted for each two stars association: comparing the time trends of the two variables across multiple countries or regions on a country or regional basis. If the trends of the two variables are similar in a significant number of countries or regions and there is a relatively stable time lag relationship between the two, this indicates a potential causal relationship, which is categorized as a three stars association. If no clear temporal relationship is observed, the association remains at the two stars level. 2.2 Data source and selection Dietary composition data in calories for 193 countries or regions for 1961-2022 from Our World in Data ( https://ourworldindata.org/grapher/dietary-composition-by-country?country=~OWID_WRL ) database was obtained, which were categorized into Alcoholic beverages, animal fats group, vegetable oils, oil crops, fish and seafood, sugar crops, sugar and sweeteners, starchy roots, other meat, sheep and goat, pig meat, poultry meat, beef, eggs, milk, nuts, fruit, vegetables, pulses, barley, maize, rice, wheat, other cereals, miscellaneous group (Share of dietary energy supplied by food commodity types in the average individual’s diet in a given country or region, measured in kilocalories per person per day). Due to missing data for some countries or regions, data for 159 countries or regions from 1992-2017 were selected in order to cover as many countries or regions as possible and over a longer period of time. All major dietary categories were further categorized into the main 16 groups: red meat, alcoholic beverages, animal fats group, milk, eggs, sugar and sweeteners, vegetable oils, nuts, poultry meat, vegetables, pulses, cereals, fish and seafood, starchy roots, fruit and other meat, of which barley, maize, rice, wheat, other cereals comprise “cereals”, pig meat, beef, sheep and goat comprise “red meat”. Other dietary components were not counted due to minimal intake and not being the mainstream dietary component. Specific information on these 16 dietary components is included in Table 1 . View this table: View inline View popup Table 1. Dietary components details Hypertension prevalence (estimated share of the population with hypertension in adults aged 30-79) for 194 countries or regions from 1990-2019 ( https://ourworldindata.org/grapher/hypertension-adults-30-79?time=latest ), diabetes prevalence (the share of people aged 20-79 who have diabetes) for 190 countries or regions from 1980-2014 ( https://ourworldindata.org/grapher/diabetes-prevalence-who-gho ), and the estimated average level of non-high-density lipoprotein cholesterol (non-HDL-C, mmol/L) for 190 countries or regions from 1980-2018 ( https://ourworldindata.org/grapher/average-non-hdl-cholesterol-levels ) were obtained from Our World in Data ( https://ourworldindata.org ) database. Likewise, to ensure that all data were relatively complete while covering a sufficiently long period of time and corresponding to the dietary group data, hypertension prevalence for 159 countries or regions from 1992-2017, diabetes prevalence from 1992-2014, average non-HDL-C levels for 159 countries or regions from 1992-2017 were selected and averaged. Gross Domestic Product (GDP) data for these 159 countries or regions from 1992-2017 were similarly obtained and averaged ( https://ourworldindata.org/grapher/gdp-per-capita-maddison-project-database ). 2.3 Statistical analysis 2.3.1 Correlation and partial correlation analysis The Pearson correlation coefficients (n = 159) between each of 16 dietary components intake per capita (1992-2017 average) and hypertension prevalence (1992-2017 average), diabetes prevalence (1992-2014 average), and average non-HDL-C levels (1992-2017 average) were calculated by SPSS (V 25), respectively. Regression analyses of each dietary component with hypertension prevalence, diabetes prevalence, and average non-HDL-C levels were performed by Prism 10 respectively. GDP per capita was taken as a control variable, and partial correlation coefficients between intake of the 16 dietary components and hypertension prevalence, diabetes prevalence, and average non-HDL-C levels were calculated using SPSS (V 25). 2.3.2 Temporal lag analysis The associations between dietary components and health indicators showed consistent significant positive or negative correlations in both one star and two stars analyses were performed temporal association analysis (three stars analysis). Eighteen countries, including China, the United States of America, and Australia, were randomly selected as examples for the analyses. The temporal trajectories of each dietary component (1961-2020) with three health indicators (hypertension prevalence, 1990-2019; diabetes prevalence, 1980-2014; and average non-HDL-C levels, 1980-2018) were examined to assess their similarity of trends. In countries or regions exhibiting a consistent trend pattern, the presence of a stable time lag between trends in dietary components and trends in health indicators was judged. For each pair of analyses, a temporal association between the dietary component and the health indicator was considered to exist if there was a clear similarity in certain countries or regions and a relatively stable time lag was present. 2.4 Data visualization Radar plots were plotted by Python to visualize the strength of one star and two stars associations between each of the 16 dietary components and hypertension prevalence, diabetes prevalence, and average non-HDL-C levels. Scatter plots with linear regression lines were plotted using Prism 10. In addition, temporal change area plots in the three stars analyses were plotted by Prism 10. Geographic heat maps of 159 countries or regions for hypertension prevalence (1992-2017 average), diabetes prevalence (1992-2014 average) and average non-HDL-C levels (1992-2017 average) were plotted with Tableau Desktop 2022.2. The graphical abstract and TSGMA flowcharts were created using bioGDP 24 ( https://biogdp.com ), and the global MA networks framework diagram was created using BioRender ( https://www.biorender.com ). 3. Results 3.1 TSGMA of the associations between dietary components and average non-HDL-C levels One star evaluation ( Figure 2a, 2b ) revealed that average non-HDL-C levels were significantly positively correlated with the intake of 11 dietary components, including red meat (Correlation Coefficient, CC = 0.566, p < 0.001), milk (CC = 0.596, p < 0.001), animal fats group (CC = 0.597, p < 0.001), sugar and sweeteners (CC = 0.676, p < 0.001), eggs (CC = 0.721, p < 0.001), fish and seafood (CC = 0.224, p < 0.001), nuts (CC = 0.316, p < 0.001), vegetable oils (CC = 0.438, p < 0.001), alcoholic beverages (CC = 0.448, p = 0.005), vegetables (CC = 0.452, p < 0.001), and poultry meat (CC = 0.492, p < 0.001). Conversely, significant negative correlations were observed for pulses (CC = −0.339, p < 0.001) and starchy roots (CC = −0.386, p < 0.001). Download figure Open in new tab Figure 2. TSGMA analysis of associations between dietary components and average non-HDL-C levels. (a) Radar plot shows one star evaluation of Pearson correlation coefficients between the intake of 16 dietary components and average non-HDL-C levels across 159 countries or regions. (b) Scatter plot illustrates the association between dietary components and average non-HDL-C levels. (c) Radar plot shows two stars evaluation after adjusting for GDP per capita, based on partial correlation coefficients. (d) Three stars analysis of temporal trends. The geographic heatmap depicts the average non-HDL-C levels (1992-2017 averages) in 159 countries or regions worldwide, with particular highlights of the 18 countries randomly selected for the temporal analysis. The area plot shows temporal trends in animal fats group intake (purple) and average non-HDL-C levels (green) from 1961 to 2020. Since GDP per capita is closely related to various life quality factors such as dietary structure, health care and education level, it can be considered as a representation of confounding factors in the global analysis 23 . Therefore, I choose the GDP per capita as an adjustment variable for the two stars analysis here. Following adjustment for GDP per capita ( Figure 2c ), average non-HDL-C levels remained significantly positively associated with 8 dietary components: alcoholic beverages (Partial Correlation Coefficients, PCC = 0.172, p = 0.041), animal fats group (PCC = 0.308, p < 0.001), vegetables (PCC = 0.317, p < 0.001), poultry meat (PCC = 0.328, p < 0.001), milk (PCC = 0.361, p < 0.001), sugar and sweeteners (PCC = 0.527, p < 0.001), eggs (PCC = 0.572, p < 0.001), and red meat (PCC = 0.259, p = 0.002). Negative associations with starchy roots (PCC = −0.361, p < 0.001) and pulses (PCC = −0.199, p = 0.018) also persisted after adjustment. Notably, associations with nuts (PCC = 0.064, p = 0.452), vegetable oils (PCC = 0.145, p = 0.084), and fish and seafood (PCC = −0.034, p = 0.692) became non-significant. Additionally, compared to the previous one star analysis, cereals (previously non-significant) showed a positive association (PCC = 0.231, p = 0.006), while other meats (previously non-significant) showed a significant negative association (PCC = −0.227, p = 0.007). Animal fats group, red meat, eggs, alcoholic beverages, milk, poultry meat, pulses, starchy roots, sugar and sweeteners and vegetables, which were already significantly associated with both one star and two stars analysis, were selected for the three stars analysis (as described in Section 2.2.1). For animal fats group, among the 18 randomly selected countries, China, the United States, Australia, Germany, the United Kingdom, Algeria, Canada, and Egypt showed similar trends between animal fats group intake and non-HDL-C levels, and changes in non-HDL-C levels lagged behind animal fats group intake changes ( Figure 2d ). For instance, in China, both animal fats group intake and non-HDL-C levels have continued to rise; in the United States, both have declined. In Algeria, animal fat group intake peaked around 1982 and began to decline thereafter, with non-HDL-C levels also peaking and declining about 16 years later. Similarly, in Egypt, animal fats group intake peaked around 1980 and non-HDL-C levels peaked around 1994 and began to decline after 2005. In summary, in countries or regions where changes in animal fats group intake follow a similar trend to that of non-HDL-C levels, there is always a lag of about 15 years. For red meat, similar lagged relationships were observed in China, the United States, Australia, the United Kingdom, Canada, Argentina, and Indonesia (Figure S1). For example, in Australia, red meat intake began declining around 1972, with non-HDL-C levels following a downward trend approximately 19 years later. In the United Kingdom, red meat intake declined from around 1971, with non-HDL-C levels decreasing about 16 years later. In Germany, a decline in red meat intake starting in 1987 preceded a fall in non-HDL-C levels by approximately six years. In Indonesia, a rise in red meat intake around 1978 was followed by an increase in non-HDL-C levels around 1996. Overall, the lag between red meat intake changes and non-HDL-C levels response ranged from 15 to 20 years. For eggs, similar lagged patterns were observed in the United States, Australia, the United Kingdom, Germany, and Canada (Figure S2). In Australia, a decline in egg consumption around 1978 was followed by a decrease in non-HDL-C levels around 1991 (lag approximately 13 years). In Germany, egg intake declined around 1980, with non-HDL-C levels falling around 1993 (lag approximately 13 years). Collectively, the lag between egg intake and non-HDL-C levels changes consistently approximated 13 years. By contrast, trends in alcoholic beverages (Figure S3), milk (Figure S4), poultry meat (Figure S5), pulses (Figure S6), starchy roots (Figure S7), sugar and sweeteners (Figure S8), and vegetables (Figure S9) were not similar to those for non-HDL-C levels in the majority of countries or regions. In summary, based on TSGMA, nuts, vegetable oils, and fish and seafood are one star risk factors for non-HDL-C levels. Alcoholic beverages, vegetables, poultry meat, milk, sugar and sweeteners are two stars risk factors for non-HDL-C levels, while starchy roots and pulses are two stars protective factors for non-HDL-C levels. Animal fats group, red meat, and eggs are three stars risk factors for non-HDL-C levels. Moreover, cereals, fruit and other meat are zero star risk factors, showing no association with non-HDL-C levels. 3.2 TSGMA of the associations between dietary components and diabetes prevalence One star evaluation ( Figure 3a, 3b ) showed that diabetes prevalence was significantly positively correlated with the intake of three dietary components: poultry meat (CC = 0.326, p < 0.001), cereals (CC = 0.287, p < 0.001), and sugar and sweeteners (CC = 0.228, p = 0.004). In contrast, significant negative correlations were observed for alcoholic beverages (CC = −0.329, p < 0.001) and starchy roots (CC = −0.238, p = 0.003). Download figure Open in new tab Figure 3. TSGMA analysis of associations between dietary components and diabetes prevalence. (a) Radar plot shows one star evaluation of Pearson correlation coefficients between the intake of 16 dietary components and diabetes prevalence across 159 countries or regions. (b) Scatter plot illustrates the association between dietary components and diabetes prevalence. (c) Radar plot shows two stars evaluation after adjusting for GDP per capita, based on partial correlation coefficients. (d) Three stars analysis of temporal trends. The geographic heatmap shows the diabetes prevalence (1992-2014 averages) in 159 countries or regions worldwide, with particular highlights of the 18 countries randomly selected for the temporal analysis. The area plot shows temporal trends in sugar and sweeteners intake (pink) and diabetes prevalence (orange) from 1961 to 2020. After adjustment for GDP per capita ( Figure 3c ), positive associations between diabetes prevalence and cereals (PCC = 0.504, p < 0.001), poultry meat (PCC = 0.380, p < 0.001), and sugar and sweeteners (PCC = 0.301, p < 0.001) remained significant. Negative associations with alcoholic beverages (PCC = −0.419, p < 0.001) and starchy roots (PCC = −0.374, p < 0.001) also persisted. Notably, vegetables, which were non-significant in one star analysis, became positively associated with diabetes after adjustment (PCC = 0.243, p = 0.004). In addition, animal fats group (PCC = –0.275, p = 0.001) and red meat (PCC = −0.223, p = 0.008) emerged as significant negative associations only after adjustment. In both the one star and two stars analysis, sugar and sweeteners, cereals, poultry meat, starchy roots, and alcoholic beverages were all consistently significantly associated with the diabetes prevalence, and they were selected for the three stars analysis. For sugar and sweeteners ( Figure 3d ), seven countries, including China, Germany, Brazil, Saudi Arabia, Mexico, Turkey, and Egypt, exhibited parallel upward trends in sugar and sweetener consumption and diabetes prevalence, with diabetes prevalence changes lagging behind diabetes prevalence intake trends. For cereals, ten countries, including China, the United States, India, South Africa, Algeria, Mexico, Indonesia, Saudi Arabia, and Egypt, demonstrated similar patterns: rising cereal intake was followed by increases in diabetes prevalence (Figure S10). Similarly, for poultry meat, the majority of the 18 randomly selected countries showed consistent upward trends in poultry meat intake and subsequent rises in diabetes prevalence, indicating a high level of temporal coherence (Figure S11). In contrast, for starchy roots, China, France, Germany, Brazil, Argentina, and Sweden displayed opposing trends between intake and diabetes prevalence, with increased consumption associated with lower diabetes prevalence (Figure S12). However, for alcoholic beverages, no consistent temporal pattern linked to diabetes prevalence was observed across the majority of countries or regions (Figure S13). Moreover, the specific value of the time lag in the three stars analysis of the relationship between diabetes prevalence and dietary components is difficult to determine because the vast majority of these countries or regions have no obvious transitions or features that can be used to quantitatively determine the lag time. Thus, in the relationship between dietary components and diabetes, poultry meat, cereals, and sugar and sweeteners are three stars risk factors for diabetes, while starchy roots are three stars protective factors. Two stars association factor include only alcoholic beverages with protective effects, no dietary components are found as one star risk factors for diabetes. Moreover, red meat, animal fats group, milk, eggs, vegetable oils, nuts, vegetables, pulses, fish and seafood, fruit and other meat are zero star risk factors, showing no association with diabetes prevalence. 3.3 TSGMA of the associations between dietary components and Hypertension One star evaluation ( Figure 4a, 4b ) showed that hypertension prevalence was significantly positively associated with milk intake (CC = 0.256, p = 0.001) and alcoholic beverage intake (CC = 0.220, p = 0.006), while a significant negative correlation was observed with fish and seafood consumption (CC = −0.194, p = 0.015). Download figure Open in new tab Figure 4. TSGMA analysis of associations between dietary components and hypertension prevalence. (a) Radar plot shows one star evaluation of Pearson correlation coefficients between the intake of 16 dietary components and average hypertension prevalence across 159 countries. (b) Scatter plot illustrates the association between dietary components and hypertension prevalence. (c) Radar plot showing two stars evaluation after adjusting for GDP per capita, based on partial correlation coefficients. (d) Three stars analysis of temporal trends. The geographic heatmap shows the hypertension prevalence (1992-2017 averages) in 159 countries worldwide, with particular highlights of the 18 countries randomly selected for the temporal analysis. The area plot shows temporal trends in alcoholic beverages intake (red) and hypertension prevalence (blue) from 1961 to 2020. Following adjustment for GDP per capita, two stars analysis ( Figure 4c ) revealed that hypertension prevalence remained significantly positively associated with milk (PCC = 0.379, p < 0.001) and alcoholic beverages (PCC = 0.296, p < 0.001). However, the previously observed negative correlation with fish and seafood intake was no longer significant (PCC = −0.144, p = 0.087). Additionally, several dietary components not associated at the one star analysis became significant after adjustment: animal fats group (PCC = 0.288, p = 0.001), eggs (PCC = 0.203, p = 0.015), and red meat (PCC = 0.195, p = 0.020) were positively correlated with hypertension prevalence, while cereals (PCC = −0.166, p = 0.049) and pulses (PCC = −0.184, p = 0.029) showed significant negative associations. In the TSGMA of dietary components and hypertension prevalence, only alcoholic beverages and milk were consecutively significantly associated with hypertension prevalence in one star and two stars analyses, thus alcoholic beverages and milk were selected for three stars analysis. For alcoholic beverages ( Figure 4d ), eight countries, including China, the United States, Australia, France, Brazil, India, Canada, and South Africa, exhibited similar temporal trends between alcoholic beverages intake and hypertension prevalence. In most cases, increases in alcoholic beverage consumption preceded subsequent rises in hypertension prevalence. For example, in China, alcoholic beverage intake rose steadily from around 1980, while hypertension prevalence showed a corresponding upward trend since before around 2010. In the United States, alcoholic beverages intake continued to rise, with a turnaround in 1980, while the turnaround in the increase of hypertension prevalence occurred around 2008, with a time lag of about 28 years. In Australia, alcoholic beverages intake showed a turnaround from declining to holding steady in 1996, while hypertension prevalence showed a similar turnaround around 2010, with a lag of about 16 years. In France, a long-term decline in alcohol consumption paralleled a decline in hypertension prevalence over the same period. In South Africa, the turnaround in the rise in alcoholic beverages intake occurred around 1984, and a similar turnaround in hypertension prevalence occurred around 2010, with a lag of approximately 26 years. By contrast, for milk, most countries or regions did not display a coherent temporal relationship between milk intake and hypertension prevalence (Figure S14). Together, these findings suggest that alcoholic beverage consumption may contribute to changes in hypertension prevalence with a temporal lag, whereas no consistent time-lagged relationship was observed for milk intake. Therefore, alcoholic beverages are three stars risk factors for hypertension, milk is a two stars risk factor, while Fish and seafood are found as a one star protective factor. However, red meat, animal fats group, eggs, sugar and sweeteners, vegetable oils, nuts, poultry meat, vegetables, pulses, cereals, starchy roots, fruit and other meat are zero star risk factors, showing no association with hypertension prevalence. 4. Discussion 4.1 Macro associations Comprehensive analysis of global-scale associations within extensive datasets is essential for understanding risk factors linked to human health 20 , 25 , 26 . The rapid expansion of data availability and computational capabilities has fundamentally altered how data are collected, processed, and utilized 27 – 29 . Traditional concepts of "big data" may no longer fully capture the scale, depth, or reliance upon contemporary datasets, which continue to expand exponentially. Here, I introduce the term “macro data” (MD) to characterize datasets that are vast in scale, multidimensional in structure, and broadly cross. In the context of MD, the association based on these data also rises to a completely new dimension, and decoding all the associations behind these data will fundamentally change our understanding of the world. Therefore, I name the association of all data behind MD as “macro association” (MA), and I name the exploration of MA as “macro association analysis” (MA-analysis). The essence of MA-analysis is to rely on MD to systematically integrate maximum possible relevant variables to reveal comprehensive multidimensional associations. Given the unprecedented volume, granularity and analytical potential of current data, MA-analysis will provide a deeper, previously unattainable understanding of the world. Extensive adoption of MA-analysis can lead to the construction of global MA networks. MA-analysis can operate at various scales, from micro-level associations such as molecular or cellular networks (comprehensive gene or protein interaction networks) to macro-level relationships involving global phenomena, including global dietary, climate, economic, or health 30 – 35 . Moreover, cross-level analyses integrating micro and macro scales through suitable MA methodologies offer additional layers of insight into complex global dynamics. Among multiple potential applications, global MA-analysis stands out due to its direct relevance to human life, health, economics, and societal dynamics. The macro-level data is particularly advantageous when compared to the intricacies of molecular-level data because of the relative simplicity of the calculations associated with processing predominantly numerical and easily interpretable data 36 , 37 . Furthermore, databases such as Our World in Data ( https://ourworldindata.org ) and growing public accessibility to global datasets have now created optimal conditions for implementing global MA-analysis. In epidemiological research, conventional correlation or regression analyses, despite their immediacy, often yield inconclusive or misleading results due to confounding factors 38 – 42 . Advanced epidemiological methodologies, including case-control studies, cohort studies, and meta-analyses, while more reliable and valid, are resource-intensive and time-consuming 42 – 44 , and most are not suitable for analyzing global data on a country-by-country basis. As a result, these rigorous methods typically focus on confirming established or controversial relationships for which the data are suitable, whereas extensive and rapid exploration of potential associations is not and equally unsuitable for MA-analysis. However, rapid, efficient and accurate analysis is essential for MA-analysis. A large amount of country-level global data is currently underutilized, despite being readily available from various sources. Therefore, there is an urgent need for streamlined, efficient, and relatively accurate methods to bridge the gap between immediate but superficial association analyses and rigorous but resource-intensive epidemiological studies, as well as to conduct and systematize MA-analysis in preparation for the establishment of global MA networks. 4.2 Three Stars Global Macro Association-analysis Given the exponential growth in global data availability, there is an urgent need for a simple, rapid, and reliable analytical approach to facilitate MA-analysis. Here, I propose the Three Stars Global Macro Association-analysis (TSGMA), a simple but accurate method designed to efficiently classify associations between risk factors through a straightforward three-step evaluation process. TSGMA categorizes associations using progressively rigorous analytical steps: correlation analysis (one star), partial correlation analysis controlling for confounders (two stars), and temporal relationship assessment for potential causality (three stars). The one star analysis initially identifies potential associations through correlation coefficients calculated from averaged national-level data across multiple years. Although straightforward, correlation analysis swiftly highlights possible association on a global scale. However, correlations alone cannot distinguish genuine associations from spurious ones influenced by confounding variables such as socioeconomic conditions, policy changes, or technological advancements 23 , 45 – 47 . Thus, further refinement is essential. For the two stars analysis, critical confounders that could significantly bias the correlation are identified and controlled via partial correlation analysis. Given that overly extensive adjustments can inadvertently mask true associations, it is crucial to select only the most significant and representative confounders to maintain accuracy 48 , 49 . Moreover, broad one star analysis can also assist in the finding of critical confounder. In brief, Variables consistently exhibiting two stars associations indicate robust relationships worthy of further consideration and investigation. In the final three stars analysis, temporal relationships between risk factors and health outcomes are assessed at the national level. Specifically, trends over time are analyzed to determine if consistent lagged relationships exist. Such temporal analyses provide stronger evidence for potential causality and further mitigate the impact of residual or unidentified confounders, enhancing the reliability of identified associations. This comprehensive and systematic method significantly advances our ability to quickly and reliably evaluate associations from extensive global datasets. TSGMA has extensive applicability, ranging from resolving longstanding debates to uncovering novel associations across diverse research fields. Meanwhile, the objects analyzed by TSGMA can be transformed according to the data conditions, either in country or regional units, and are applicable to issues with similar data conditions. The multiple levels also allow variables with different degrees of association to be revealed, rather than a single criterion causing the omission of numerous weak or moderate associations. Moreover, a harmonized and standardized approach to quantifying these associations also makes them comparable. As an example, the TSGMA system I conducted in this study revealed associations between global dietary patterns and three major cardiometabolic health indicators (Non-HDL-C levels, diabetes prevalence, and hypertension prevalence). Among numerous dietary factors, animal fat, red meat, and eggs emerged as robust three stars risk factors strongly associated with elevated non-HDL-C levels and with lag times ranging from approximately 13 to 20 years. Notably, these findings are highly consistent with established evidence in nutritional epidemiology 50 – 52 . In contrast, pulses and starchy roots steadily demonstrated protective two stars associations, reflecting their role in lipid regulation due to their high fiber content and lower glycemic impact 53 , 54 . In the context of diabetes, poultry meat, cereals, and sugar and sweeteners exhibited strong, persistent three stars associations, aligning well with existing evidence highlighting refined carbohydrates and excessive poultry intake as potential diabetes risk factors through mechanisms involving insulin resistance and impaired glucose homeostasis 55 – 57 . Conversely, Starchy roots demonstrated a protective two stars association, as did with non-HDL-C levels 58 . In terms of hypertension prevalence, intake of alcoholic beverages showed a three stars risk association with hypertension. This strong association highlights the physiological pathways by which alcohol elevates blood pressure through mechanisms such as vasoconstriction and sympathetic activation 59 , 60 . Milk intake presented a two stars association, but lacked consistent temporal trends across most analyzed countries or regions, suggesting differential regional consumption patterns 61 . Interestingly, fish and seafood intake were identified as a one stars protective factor, aligning with prior literature that suggests a clear cardiovascular protective effect of omega-3 fatty acids from fish oil 62 , 63 . However, the association did not reach a higher level of robustness may due to the infrequent, geographically-specific, and idiosyncratic nature of fish and seafood consumption, as has been observed in previous studies indicating the instability of this association 62 , 64 , 65 . Moreover, I and Qi successfully applied a similar approach to systematically reassess the long-debated relationship between red meat consumption and cancer incidence previously 23 , providing robust conclusions from a novel analytical perspective. Taken together, these findings underscore TSGMA’s utility in rapidly identifying dietary risk factors at a global scale, confirming known associations, and uncovering novel patterns that traditional epidemiological studies might miss due to methodological constraints. These results provide actionable insights for public health interventions targeting dietary modifications, highlighting areas needing further mechanistic exploration to substantiate causality and inform global dietary guidelines. Critically, TSGMA lays the foundation for future MA-analyses and the construction of comprehensive global MA networks by providing accurate assessments of association strength, clearly defining different levels of risk, and dramatically reducing computational and resource demands without sacrificing analytical rigor or accuracy. 4.3 Construction of global MA networks TSGMA can be effectively employed using existing data frameworks organized by either specific research domains or databases. In the case of databases, currently, most global databases primarily function as repositories for data collection rather than platforms for integrated analytical insight 66 – 68 . By introducing MA-analysis into public databases such as Our World in Data ( https://ourworldindata.org ), FAO ( https://www.fao.org/faostat/en/#data/FBS ), International Agency for Research on Cancer ( https://www.iarc.who.int ), and the World Bank ( https://data.worldbank.org ) or self-constructed datasets, we can systematically interconnect collected variables through TSGMA. Subsequent analyses utilizing artificial intelligence and advanced visualization techniques can facilitate the construction of intra-database MA networks. This transformative approach moves beyond traditional data aggregation towards integrated data analysis and visualization, representing a significant evolution in database functionality 66 – 68 . As individual databases progressively establish their internal MA networks, inter-database integration can occur, linking networks across different data repositories. This interconnection allows for cross-validation of overlapping data and addresses gaps through complementary datasets, ultimately culminating in a unified global MA network. Moreover, these interconnected MA networks can serve as the foundation for developing a dedicated global MA networks database. Such a resource would systematically document and continuously update one star, two stars, and three stars associations among global data variables, profoundly enhancing our understanding of global dynamics and facilitating solutions to complex challenges such as climate change, disease epidemiology, economic fluctuations, and environmental shifts. Each analytical level within TSGMA (from one to three stars) holds an important value. Even initial one star assessments, identifying potential correlations without confirmed causality, can offer significant insights into global patterns. This has important insights for the initial phase of MA-network construction. For example, constructing a one star MA network using Our World in Data ( https://ourworldindata.org ) could rapidly enhance our understanding of global relationships. Progressively incorporating two to three stars analyses and considering graded correlation strengths will further enrich these networks. Looking ahead, a global MA-analysis initiative could systematically apply MA-analysis across all fields. I propose a structured framework for constructing MA networks ( Figure 5 ): starting from TSGMA within individual databases or domains, progressing through inter-database or inter-domain networks, and culminating in comprehensive global MA networks. Ultimately, by integrating macro-level data (such as global socio-economic indicators) with micro-level biological data (such as proteins, nucleic acids), we can establish fully integrated domain-wide MA networks. This approach promises unprecedented insights into the interconnected dynamics shaping our world. Download figure Open in new tab Figure 5. Global MA networks development framework 5. Conclusion In this study, I introduce the concept of MA for the first time and highlight the urgency and importance of conducting MA-analysis in the context of rapidly expanding MD. To address this need, I developed the TSGMA, a novel framework designed to identify MA within increasingly available harmonized global datasets. TSGMA integrates correlation analysis, confounder adjustment, and temporal validation to efficiently and accurately classify associations into a standardized, graded system. When applied to global dietary and health indicator data, TSGMA successfully identified both well-established and previously overlooked associations, demonstrating its capability to extract actionable insights from complex global data. Beyond its analytical function, TSGMA also provides a structural foundation for building MA networks. Leveraging both TSGMA and MA-analysis, I propose a framework for constructing global MA networks: starting from intra-database networks and progressing toward the integration of inter-database associations. This framework offers a unified platform for tracking, comparing, and interpreting global drivers of health, economy, and environment, etc. Together, the integration of MA, TSGMA, and global MA networks establishes a structured and scalable approach to uncovering meaningful patterns from MD, providing a practical foundation for systematic discovery and hypothesis generation in global research. Limitations of the study Despite its advantages, TSGMA has several limitations. First, precise adjustment for confounders remains challenging, requiring more systematic methods such as multivariate regression or principal component analysis. Second, variables initially non-significant in one star analysis but becoming significant after adjusting for confounders may provide overlooked insights. Third, temporal validation (three stars analysis) needs refinement; integrating machine learning could enhance accuracy in detecting time-lag relationships. Moreover, to facilitate rapid and broad screening, I assessed associations between individual factors and outcomes without considering their interactions. Although major confounders were controlled, potential inaccuracies from correlated factors remain; for instance, dietary patterns often link meat intake with alcohol consumption, an interaction were also observed in this study. Future analyses should further incorporate such interactive or cumulative effects. Finally, integrating automated artificial intelligence approaches with TSGMA is urgently needed to further enhance the efficiency of MA-analysis. Data Availability All data produced are available online at [Database: https://ourworldindata.org ] RESOURCE AVAILABILITY Lead contact Further information and reasonable requests for resources and reagents should be directed to and will be fulfilled by the lead contact, Hongyue Ma ( mahongyue_edu{at}126.com ). Materials availability This study did not generate new samples or unique reagents. Data and code availability • Data : This paper analyzes existing, publicly available data, accessible at [Database: https://ourworldindata.org ]. • Code : This paper does not report original code. • Additional Information : Any additional information required to reanalyze the data reported in this article is available from the lead contact upon request. AUTHOR CONTRIBUTIONS Hongyue Ma performed all analyses and wrote the manuscript. DECLARATION OF INTERESTS The author declares no competing interests. ACKNOWLEDGMENTS This work is mainly supported by Hongyue Ma’s personal funds and resources. Footnotes ↵ 3 Lead contact References 1. ↵ Global burden of 87 risk factors in 204 countries and territories, 1990-2019: a systematic analysis for the Global Burden of Disease Study 2019 . ( 2020 ). Lancet 396 , 1223 – 1249 . doi: 10.1016/s0140-6736(20)30752-2 . OpenUrl CrossRef PubMed 2. Willett , W. , Rockström , J. , Loken , B. , Springmann , M. , Lang , T. , Vermeulen , S. , Garnett , T. , Tilman , D. , DeClerck , F. , Wood , A. , et al. ( 2019 ). Food in the Anthropocene: the EAT-Lancet Commission on healthy diets from sustainable food systems . Lancet 393 , 447 – 492 . doi: 10.1016/s0140-6736(18)31788-4 . OpenUrl CrossRef 3. ↵ Czeisler , C.A . ( 2013 ). Perspective: casting light on sleep deficiency . Nature 497 , S13 . doi: 10.1038/497S13a . OpenUrl CrossRef PubMed Web of Science 4. ↵ Susser , M . ( 1994 ). The logic in ecological: I. The logic of analysis . Am J Public Health 84 , 825 – 829 . doi: 10.2105/ajph.84.5.825 . OpenUrl CrossRef PubMed Web of Science 5. ↵ Doll , R. , and Hill , A.B . ( 1950 ). Smoking and carcinoma of the lung; preliminary report . Br Med J 2 , 739 – 748 . doi: 10.1136/bmj.2.4682.739 . OpenUrl FREE Full Text 6. ↵ Ioannidis , J.P . ( 2005 ). Why most published research findings are false . PLoS Med 2 , e124 . doi: 10.1371/journal.pmed.0020124 . OpenUrl CrossRef PubMed 7. Mills , A . ( 2014 ). Health care systems in low- and middle-income countries . N Engl J Med 370 , 552 – 557 . doi: 10.1056/NEJMra1110897 . OpenUrl CrossRef PubMed 8. ↵ Greenland , S. , and Robins , J . ( 1994 ). Invited commentary: ecologic studies--biases, misconceptions, and counterexamples . Am J Epidemiol 139 , 747 – 760 . doi: 10.1093/oxfordjournals.aje.a117069 . OpenUrl CrossRef PubMed Web of Science 9. ↵ Collins , R. , and MacMahon , S . ( 2001 ). Reliable assessment of the effects of treatment on mortality and major morbidity, I: clinical trials . Lancet 357 , 373 – 380 . doi: 10.1016/s0140-6736(00)03651-5 . OpenUrl CrossRef PubMed Web of Science 10. ↵ Sudlow , C. , Gallacher , J. , Allen , N. , Beral , V. , Burton , P. , Danesh , J. , Downey , P. , Elliott , P. , Green , J. , Landray , M. , et al. ( 2015 ). UK biobank: an open access resource for identifying the causes of a wide range of complex diseases of middle and old age . PLoS Med 12 , e1001779 . doi: 10.1371/journal.pmed.1001779 . OpenUrl CrossRef PubMed 11. ↵ Ioannidis , J.P . ( 2016 ). The Mass Production of Redundant, Misleading, and Conflicted Systematic Reviews and Meta-analyses . Milbank Q 94 , 485 – 514 . doi: 10.1111/1468-0009.12210 . OpenUrl CrossRef PubMed 12. ↵ Ioannidis , J.P . ( 2016 ). Why Most Clinical Research Is Not Useful . PLoS Med 13 , e1002049 . doi: 10.1371/journal.pmed.1002049 . OpenUrl CrossRef PubMed 13. Manolio , T.A. , Weis , B.K. , Cowie , C.C. , Hoover , R.N. , Hudson , K. , Kramer , B.S. , Berg , C. , Collins , R. , Ewart , W. , Gaziano , J.M. , et al. ( 2012 ). New models for large prospective studies: is there a better way? Am J Epidemiol 175 , 859 – 866 . doi: 10.1093/aje/kwr453 . OpenUrl CrossRef PubMed Web of Science 14. ↵ Batty , G.D. , Gale , C.R. , Kivimäki , M. , Deary , I.J. , and Bell , S . ( 2020 ). Comparison of risk factor associations in UK Biobank against representative, general population based studies with conventional response rates: prospective cohort study and individual participant meta-analysis . Bmj 368 , m131 . doi: 10.1136/bmj.m131 . OpenUrl Abstract / FREE Full Text 15. ↵ Lazer , D. , Pentland , A. , Adamic , L. , Aral , S. , Barabasi , A.L. , Brewer , D. , Christakis , N. , Contractor , N. , Fowler , J. , Gutmann , M. , et al. ( 2009 ). Social science. Computational social science . Science 323 , 721 – 723 . doi: 10.1126/science.1167742 . OpenUrl Abstract / FREE Full Text 16. Pisani , E. , Aaby , P. , Breugelmans , J.G. , Carr , D. , Groves , T. , Helinski , M. , Kamuya , D. , Kern , S. , Littler , K. , Marsh , V. , et al. ( 2016 ). Beyond open data: realising the health benefits of sharing data . BMJ 355 , i5295 . doi: 10.1136/bmj.i5295 . OpenUrl FREE Full Text 17. ↵ Leonelli , S . ( 2014 ). What Difference Does Quantity Make? On the Epistemology of Big Data in Biology . Big Data Soc 1 . doi: 10.1177/2053951714534395 . OpenUrl CrossRef 18. Krieger , N . ( 1994 ). Epidemiology and the web of causation: has anyone seen the spider? Soc Sci Med 39 , 887 – 903 . doi: 10.1016/0277-9536(94)90202-x . OpenUrl CrossRef PubMed Web of Science 19. ↵ Diez Roux , A.V. ( 2011 ). Complex systems thinking and current impasses in health disparities research . Am J Public Health 101 , 1627 – 1634 . doi: 10.2105/ajph.2011.300149 . OpenUrl CrossRef PubMed Web of Science 20. ↵ Keyes , K. , and Galea , S . ( 2015 ). What matters most: quantifying an epidemiology of consequence . Ann Epidemiol 25 , 305 – 311 . doi: 10.1016/j.annepidem.2015.01.016 . OpenUrl CrossRef PubMed 21. ↵ Pearce , N . ( 1996 ). Traditional epidemiology, modern epidemiology, and public health . Am J Public Health 86 , 678 – 683 . doi: 10.2105/ajph.86.5.678 . OpenUrl CrossRef PubMed Web of Science 22. ↵ Sterne , J.A. , Hernán , M.A. , Reeves , B.C. , Savović , J. , Berkman , N.D. , Viswanathan , M. , Henry , D. , Altman , D.G. , Ansari , M.T. , Boutron , I. , et al. ( 2016 ). ROBINS-I: a tool for assessing risk of bias in non-randomised studies of interventions . Bmj 355 , i4919 . doi: 10.1136/bmj.i4919 . OpenUrl FREE Full Text 23. ↵ Ma , H. , and Qi , X . ( 2023 ). Red Meat Consumption and Cancer Risk: A Systematic Analysis of Global Data . Foods 12 . doi: 10.3390/foods12224164 . OpenUrl CrossRef 24. ↵ Jiang , S. , Li , H. , Zhang , L. , Mu , W. , Zhang , Y. , Chen , T. , Wu , J. , Tang , H. , Zheng , S. , Liu , Y. , et al. ( 2025 ). Generic Diagramming Platform (GDP): a comprehensive database of high-quality biomedical graphics . Nucleic Acids Res 53 , D1670 – d1676 . doi: 10.1093/nar/gkae973 . OpenUrl CrossRef PubMed 25. ↵ Murray , C.J.L. , and Lopez , A.D . ( 2017 ). Measuring global health: motivation and evolution of the Global Burden of Disease Study . Lancet 390 , 1460 – 1464 . doi: 10.1016/s0140-6736(17)32367-x . OpenUrl CrossRef PubMed 26. ↵ Rutter , H. , Savona , N. , Glonti , K. , Bibby , J. , Cummins , S. , Finegood , D.T. , Greaves , F. , Harper , L. , Hawe , P. , Moore , L. , et al. ( 2017 ). The need for a complex systems model of evidence for public health . Lancet 390 , 2602 – 2604 . doi: 10.1016/s0140-6736(17)31267-9 . OpenUrl CrossRef 27. ↵ Khoury , M.J. , and Ioannidis , J.P . ( 2014 ). Medicine. Big data meets public health . Science 346 , 1054 – 1055 . doi: 10.1126/science.aaa2709 . OpenUrl Abstract / FREE Full Text 28. Emmert-Streib , F . ( 2021 ). From the digital data revolution toward a digital society: Pervasiveness of artificial intelligence . Machine Learning and Knowledge Extraction 3 , 284 – 298 . OpenUrl 29. ↵ Cao , L . ( 2017 ). Data science: a comprehensive overview . ACM Computing Surveys (CSUR ) 50 , 1 – 42 . OpenUrl 30. ↵ Barabási , A.L. , and Oltvai , Z.N . ( 2004 ). Network biology: understanding the cell’s functional organization . Nat Rev Genet 5 , 101 – 113 . doi: 10.1038/nrg1272 . OpenUrl CrossRef PubMed Web of Science 31. Menichetti , G. , Barabási , A.L. , and Loscalzo , J . ( 2024 ). Decoding the Foodome: Molecular Networks Connecting Diet and Health . Annu Rev Nutr 44 , 257 – 288 . doi: 10.1146/annurev-nutr-062322-030557 . OpenUrl CrossRef PubMed 32. Donges , J.F. , Zou , Y. , Marwan , N. , and Kurths , J . ( 2009 ). Complex networks in climate dynamics . European Physical Journal Special Topics 174 , 157 – 179 . OpenUrl 33. Steinhaeuser , K. , Chawla , N.V. , and Ganguly , A.R . ( 2011 ). Complex networks as a unified framework for descriptive analysis and predictive modeling in climate science . Statistical Analysis and Data Mining . 34. Ghiassian , S.D . ( 2015 ). Network Medicine: A Network-based Approach to Human Diseases . Dissertations & Theses - Gradworks . 35. ↵ Sonawane , A.R. , Weiss , S.T. , Glass , K. , and Sharma , A . ( 2019 ). Network Medicine in the age of biomedical big data . Frontiers in Genetics 10 . 36. ↵ Aiello , L.M. , Schifanella , R. , Quercia , D. , and Prete , L.D . ( 2019 ). Large-scale and high-resolution analysis of food purchases and health outcomes . EPJ Data Science 8 . 37. ↵ Haw , D.J. , Morgenstern , C. , Forchini , G. , Johnson , R. , Doohan , P. , Smith , P.C. , and Hauck , K.D . ( 2022 ). Data needs for integrated economic-epidemiological models of pandemic mitigation policies . Epidemics 41 , 100644 . OpenUrl PubMed 38. ↵ Keiding , N. , and Clayton , D . ( 2014 ). Standardization and Control for Confounding in Observational Studies: A Historical Perspective . Statistical science 29 , 529 – 558 . OpenUrl CrossRef 39. D’Amico , F. , Marmiere , M. , Fonti , M. , Battaglia , M. , and Belletti , A . ( 2025 ). Association Does Not Mean Causation, When Observational Data Were Misinterpreted as Causal: The Observational Interpretation Fallacy . J Eval Clin Pract 31 , e14288 . doi: 10.1111/jep.14288 . OpenUrl CrossRef PubMed 40. Flanders , W.D. , Strickland , M.J. , and Klein , M . ( 2017 ). A New Method for Partial Correction of Residual Confounding in Time-Series and Other Observational Studies . Am J Epidemiol 185 , 941 – 949 . doi: 10.1093/aje/kwx013 . OpenUrl CrossRef PubMed 41. Luque-Fernandez , M.A. , Schomaker , M. , Redondo-Sanchez , D. , Jose Sanchez Perez , M. , Vaidya , A. , and Schnitzer , M.E. ( 2019 ). Educational Note: Paradoxical collider effect in the analysis of non-communicable disease epidemiological data: a reproducible illustration and web application . Int J Epidemiol 48 , 640 – 653 . doi: 10.1093/ije/dyy275 . OpenUrl CrossRef 42. ↵ Shaw , P.A. , Deffner , V. , Keogh , R.H. , Tooze , J.A. , Dodd , K.W. , Küchenhoff , H. , Kipnis , V. , and Freedman , L.S . ( 2018 ). Epidemiologic analyses with error-prone exposures: review of current practice and recommendations . Ann Epidemiol 28 , 821 – 828 . doi: 10.1016/j.annepidem.2018.09.001 . OpenUrl CrossRef 43. Coduras , A. , Rabasa , I. , Frank , A. , Bermejo-Pareja , F. , López-Pousa , S. , López-Arrieta , J.M. , Del Llano , J. , León , T. , and Rejas , J . ( 2010 ). Prospective one-year cost-of-illness study in a cohort of patients with dementia of Alzheimer’s disease type in Spain: the ECO study . J Alzheimers Dis 19 , 601 – 615 . doi: 10.3233/jad-2010-1258 . OpenUrl CrossRef PubMed 44. ↵ Blettner , M. , Krahn , U. , and Schlattmann , P. ( 2014 ). Meta-Analysis in Epidemiology . In Handbook of Epidemiology , W. Ahrens , and I. Pigeot , eds. ( Springer New York ), pp. 1377 – 1411 . doi: 10.1007/978-0-387-09834-0_21 . OpenUrl CrossRef 45. ↵ Capili , B. , and Anastasi , J.K . ( 2023 ). Improving the Validity of Causal Inferences in Observational Studies . Am J Nurs 123 , 45 – 49 . doi: 10.1097/01.Naj.0000911536.51764.47 . OpenUrl CrossRef PubMed 46. Lee , Y. , and Ogburn , E.L . ( 2020 ). Network Dependence Can Lead to Spurious Associations and Invalid Inference . Journal of the American Statistical Association , 1 – 31 . 47. ↵ Sassenhagen , J. , and Alday , P.M . ( 2016 ). A common misapplication of statistical inference: Nuisance control with null-hypothesis significance tests . Brain Lang 162 , 42 – 45 . doi: 10.1016/j.bandl.2016.08.001 . OpenUrl CrossRef PubMed 48. ↵ Schisterman , E.F. , Cole , S.R. , and Platt , R.W . ( 2009 ). Overadjustment bias and unnecessary adjustment in epidemiologic studies . Epidemiology 20 , 488 – 495 . doi: 10.1097/EDE.0b013e3181a819a1 . OpenUrl CrossRef PubMed Web of Science 49. ↵ Gao , Y. , Xiang , L. , Yi , H. , Song , J. , Sun , D. , Xu , B. , Zhang , G. , and Wu , I.X . ( 2025 ). Confounder adjustment in observational studies investigating multiple risk factors: a methodological study . BMC Med 23 , 132 . doi: 10.1186/s12916-025-03957-8 . OpenUrl CrossRef PubMed 50. ↵ Mensink , R.P. , Zock , P.L. , Kester , A.D. , and Katan , M.B . ( 2003 ). Effects of dietary fatty acids and carbohydrates on the ratio of serum total to HDL cholesterol and on serum lipids and apolipoproteins: a meta-analysis of 60 controlled trials . Am J Clin Nutr 77 , 1146 – 1155 . doi: 10.1093/ajcn/77.5.1146 . OpenUrl Abstract / FREE Full Text 51. Spence , J.D. , Srichaikul , K.K. , and Jenkins , D.J.A . ( 2021 ). Cardiovascular Harm From Egg Yolk and Meat: More Than Just Cholesterol and Saturated Fat . J Am Heart Assoc 10 , e017066 . doi: 10.1161/jaha.120.017066 . OpenUrl CrossRef PubMed 52. ↵ Sacks , F.M. , Lichtenstein , A.H. , Wu , J.H.Y. , Appel , L.J. , Creager , M.A. , Kris-Etherton , P.M. , Miller , M. , Rimm , E.B. , Rudel , L.L. , Robinson , J.G. , et al. ( 2017 ). Dietary Fats and Cardiovascular Disease: A Presidential Advisory From the American Heart Association . Circulation 136 , e1 – e23 . doi: 10.1161/cir.0000000000000510 . OpenUrl Abstract / FREE Full Text 53. ↵ Tsitsou , S. , Athanasaki , C. , Dimitriadis , G. , and Papakonstantinou , E . ( 2023 ). Acute Effects of Dietary Fiber in Starchy Foods on Glycemic and Insulinemic Responses: A Systematic Review of Randomized Controlled Crossover Trials . Nutrients 15 . doi: 10.3390/nu15102383 . OpenUrl CrossRef 54. ↵ Mizelman , E. , Chilibeck , P.D. , Hanifi , A. , Kaviani , M. , Brenna , E. , and Zello , G.A . ( 2020 ). A Low-Glycemic Index, High-Fiber, Pulse-Based Diet Improves Lipid Profile, but Does Not Affect Performance in Soccer Players . Nutrients 12 . doi: 10.3390/nu12051324 . OpenUrl CrossRef 55. ↵ Hosseini-Esfahani , F. , Beheshti , N. , Koochakpoor , G. , Mirmiran , P. , and Azizi , F . ( 2022 ). Meat Food Group Intakes and the Risk of Type 2 Diabetes Incidence . Front Nutr 9 , 891111 . doi: 10.3389/fnut.2022.891111 . OpenUrl CrossRef PubMed 56. Gross , L.S. , Li , L. , Ford , E.S. , and Liu , S . ( 2004 ). Increased consumption of refined carbohydrates and the epidemic of type 2 diabetes in the United States: an ecologic assessment . Am J Clin Nutr 79 , 774 – 779 . doi: 10.1093/ajcn/79.5.774 . OpenUrl Abstract / FREE Full Text 57. ↵ Debras , C. , Deschasaux-Tanguy , M. , Chazelas , E. , Sellem , L. , Druesne-Pecollo , N. , Esseddik , Y. , Szabo de Edelenyi , F. , Agaësse , C. , De Sa , A. , Lutchia , R. , et al. ( 2023 ). Artificial Sweeteners and Risk of Type 2 Diabetes in the Prospective NutriNet-Santé Cohort . Diabetes Care 46 , 1681 – 1690 . doi: 10.2337/dc23-0206 . OpenUrl CrossRef 58. ↵ Chiavaroli , L. , Lee , D. , Ahmed , A. , Cheung , A. , Khan , T.A. , Blanco , S. , Mejia , Mirrahimi , A. , Jenkins , D.J.A. , Livesey , G. , et al. ( 2021 ). Effect of low glycaemic index or load dietary patterns on glycaemic control and cardiometabolic risk factors in diabetes: systematic review and meta-analysis of randomised controlled trials . Bmj 374 , n1651 . doi: 10.1136/bmj.n1651 . OpenUrl Abstract / FREE Full Text 59. ↵ Hering , D. , Kucharska , W. , Kara , T. , Somers , V.K. , and Narkiewicz , K . ( 2011 ). Potentiated sympathetic and hemodynamic responses to alcohol in hypertensive vs. normotensive individuals . J Hypertens 29 , 537 – 541 . doi: 10.1097/HJH.0b013e328342b2a9 . OpenUrl CrossRef PubMed 60. ↵ Bigalke , J.A. , Greenlund , I.M. , Solis-Montenegro , T.X. , Durocher , J.J. , Joyner , M.J. , and Carter , J.R . ( 2024 ). Binge Alcohol Consumption Elevates Sympathetic Transduction to Blood Pressure: A Randomized Controlled Trial . Hypertension 81 , 2140 – 2151 . doi: 10.1161/hypertensionaha.124.23416 . OpenUrl CrossRef 61. ↵ Singh , G.M. , Micha , R. , Khatibzadeh , S. , Shi , P. , Lim , S. , Andrews , K.G. , Engell , R.E. , Ezzati , M. , and Mozaffarian , D . ( 2015 ). Global, Regional, and National Consumption of Sugar-Sweetened Beverages, Fruit Juices, and Milk: A Systematic Assessment of Beverage Intake in 187 Countries . PLoS One 10 , e0124845 . doi: 10.1371/journal.pone.0124845 . OpenUrl CrossRef PubMed 62. ↵ Abdelhamid , A.S. , Brown , T.J. , Brainard , J.S. , Biswas , P. , Thorpe , G.C. , Moore , H.J. , Deane , K.H. , AlAbdulghafoor , F.K. , Summerbell , C.D. , Worthington , H.V. , et al. ( 2018 ). Omega-3 fatty acids for the primary and secondary prevention of cardiovascular disease . Cochrane Database Syst Rev 11 , Cd003177 . doi: 10.1002/14651858.CD003177.pub4 . OpenUrl CrossRef PubMed 63. ↵ George , M. , and Gupta , A . ( 2022 ). Blood Pressure-Lowering Effects of Omega-3 Polyunsaturated Fatty Acids: Are These the Missing Link to Explain the Relationship Between Omega-3 Polyunsaturated Fatty Acids and Cardiovascular Disease? J Am Heart Assoc 11 , e026258 . doi: 10.1161/jaha.121.026258 . OpenUrl CrossRef PubMed 64. ↵ Nahab , F. , Le , A. , Judd , S. , Frankel , M.R. , Ard , J. , Newby , P.K. , and Howard , V.J. ( 2011 ). Racial and geographic differences in fish consumption: the REGARDS study . Neurology 76 , 154 – 158 . doi: 10.1212/WNL.0b013e3182061afb . OpenUrl CrossRef PubMed 65. ↵ Matsumoto , C. , Yoruk , A. , Wang , L. , Gaziano , J.M. , and Sesso , H.D . ( 2019 ). Fish and omega-3 fatty acid consumption and risk of hypertension . J Hypertens 37 , 1223 – 1229 . doi: 10.1097/hjh.0000000000002062 . OpenUrl CrossRef PubMed 66. ↵ Kekevi , U. , and Aydin , A . ( 2022 ). Real-Time Big Data Processing and Analytics: Concepts , Technologies, and Domains. Computer Science . 67. Horvitz , E. , and Mitchell , T. ( 2020 ). From Data to Knowledge to Action: A Global Enabler for the 21st Century . 68. ↵ Chapman , A. , Simperl , E. , Koesten , L. , Konstantinidis , G. , and Groth , P . ( 2020 ). Dataset search: a survey . The VLDB Journal 29 . View the discussion thread. Back to top Previous Next Posted June 11, 2025. Download PDF Supplementary Material Data/Code Email Thank you for your interest in spreading the word about medRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following TSGMA: identification of macro associations from global data to build global MA networks Message Subject (Your Name) has forwarded a page to you from medRxiv Message Body (Your Name) thought you would like to see this page from the medRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share TSGMA: identification of macro associations from global data to build global MA networks Hongyue Ma medRxiv 2025.06.09.25329257; doi: https://doi.org/10.1101/2025.06.09.25329257 Share This Article: Copy Citation Tools TSGMA: identification of macro associations from global data to build global MA networks Hongyue Ma medRxiv 2025.06.09.25329257; doi: https://doi.org/10.1101/2025.06.09.25329257 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Epidemiology Subject Areas All Articles Addiction Medicine (568) Allergy and Immunology (863) Anesthesia (300) Cardiovascular Medicine (4435) Dentistry and Oral Medicine (444) Dermatology (382) Emergency Medicine (608) Endocrinology (including Diabetes Mellitus and Metabolic Disease) (1509) Epidemiology (15229) Forensic Medicine (30) Gastroenterology (1124) Genetic and Genomic Medicine (6600) Geriatric Medicine (668) Health Economics (997) Health Informatics (4536) Health Policy (1368) Health Systems and Quality Improvement (1613) Hematology (541) HIV/AIDS (1264) Infectious Diseases (except HIV/AIDS) (15916) Intensive Care and Critical Care Medicine (1103) Medical Education (623) Medical Ethics (146) Nephrology (667) Neurology (6599) Nursing (346) Nutrition (998) Obstetrics and Gynecology (1144) Occupational and Environmental Health (957) Oncology (3332) Ophthalmology (974) Orthopedics (369) Otolaryngology (420) Pain Medicine (436) Palliative Medicine (130) Pathology (663) Pediatrics (1693) Pharmacology and Therapeutics (691) Primary Care Research (711) Psychiatry and Clinical Psychology (5447) Public and Global Health (9232) Radiology and Imaging (2198) Rehabilitation Medicine and Physical Therapy (1370) Respiratory Medicine (1196) Rheumatology (593) Sexual and Reproductive Health (712) Sports Medicine (530) Surgery (712) Toxicology (99) Transplantation (289) Urology (265) (function(){function c(){var b=a.contentDocument||a.contentWindow.document;if(b){var d=b.createElement('script');d.innerHTML="window.__CF$cv$params={r:'a00b9f38ef670db4',t:'MTc3OTYxODU5NA=='};var a=document.createElement('script');a.src='/cdn-cgi/challenge-platform/scripts/jsd/main.js';document.getElementsByTagName('head')[0].appendChild(a);";b.getElementsByTagName('head')[0].appendChild(d)}}if(document.body){var a=document.createElement('iframe');a.height=1;a.width=1;a.style.position='absolute';a.style.top=0;a.style.left=0;a.style.border='none';a.style.visibility='hidden';document.body.appendChild(a);if('loading'!==document.readyState)c();else if(window.addEventListener)document.addEventListener('DOMContentLoaded',c);else{var e=document.onreadystatechange||function(){};document.onreadystatechange=function(b){e(b);'loading'!==document.readyState&&(document.onreadystatechange=e,c())}}}})();

Text is read by the "Ask this paper" AI Q&A widget below. Extraction quality varies by source — PMC NXML preserves structure cleanly, OA-HTML may include some navigation residue, and OA-PDF can have broken hyphenation. The publisher copy (via DOI) is the canonical version.

My notes (saved in your browser only)

Ask this paper AI returns verbatim quotes from the full text · source: preprint-html

Answers must be backed by verbatim quotes from this paper's full text. Hallucinated quotes are dropped automatically; if no verbatim passage answers the question, we say so. How this works

Citation neighborhood (no data yet)

We don't have any in-corpus citations linked to this paper yet. This is a recent paper (2025) — citers typically take a year or two to land, and the OpenAlex reference graph may still be filling in.

Source provenance

europepmc
last seen: 2026-05-20T01:45:00.602351+00:00