Full text
43,569 characters
· extracted from
preprint-html
· click to expand
Investigating the demographic history of Sindhi population inhabited in West coast India | bioRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-M677548'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search New Results Investigating the demographic history of Sindhi population inhabited in West coast India View ORCID Profile Lomous Kumar , Suraj Nongmaithem , Sachin Kumar , View ORCID Profile Kumarasamy Thangaraj doi: https://doi.org/10.1101/2025.03.01.640946 Lomous Kumar 1 CSIR-Centre for Cellular and Molecular Biology , Hyderabad 500007, India 2 Birbal Sahni Institute of Palaeosciences , Lucknow 226007, India Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Lomous Kumar For correspondence: thangs{at}ccmb.res.in lomousmishra{at}gmail.com Suraj Nongmaithem 1 CSIR-Centre for Cellular and Molecular Biology , Hyderabad 500007, India Find this author on Google Scholar Find this author on PubMed Search for this author on this site Sachin Kumar 2 Birbal Sahni Institute of Palaeosciences , Lucknow 226007, India Find this author on Google Scholar Find this author on PubMed Search for this author on this site Kumarasamy Thangaraj 1 CSIR-Centre for Cellular and Molecular Biology , Hyderabad 500007, India 3 Tata Institute for Genetics and Society , Bangalore 560065, India Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Kumarasamy Thangaraj For correspondence: thangs{at}ccmb.res.in lomousmishra{at}gmail.com Abstract Full Text Info/History Metrics Supplementary material Preview PDF Abstract Background South Asian populations are genetically well stratified due to multiple waves of migration, admixture events, and endogamy. India remains a rich resource for population genomics studies with many small and socioculturally homogeneous communities whose origins and demographic histories are largely unknown. In this study, we analysed such a small Sindhi settlement in the Thane district in Maharashtra of West coast India using genome-wide autosomal SNP data using both frequency- and haplotype-based approaches. Results Our analyses suggest that the West coast Indian Sindhi community is very unique and has significant population affinity with a group more closely related to the Pakistani Burusho than to the Pakistani Sindhi, as it has an additional East/Southeast Asian component. Furthermore, the sharing of haplotype and IBD suggests recent gene flow from the local Konkani population on the west coast of India into Indian Sindhi. Admixture modelling suggested that Indian Sindhi admixture with the East/Southeast Asian source group could be 40-50 GBP, explaining their current unique demographics. However, apart from this additional admixture, they share the basic genetic composition of the Pakistan/NWI groups, as reflected in PCA, outgroup F3 and IBD sharing. Conclusion These new findings suggest that Indian Sindhi settlement from the Thane in Maharashtra in West coast of India derive their genetic ancestry not directly from Pakistani Sindhis but from other groups related to Burusho in Pakistan. The study therefore encourages further research to identify the heterogeneous nature of migrations to the Indian subcontinent and thus further decipher its unique demographics. Introduction South Asia is considered as one of the most geographically diverse locations in terms of linguistics, culture, and genetics ( 1 , 2 ). Among the South Asian countries, India and Pakistan are the land of fascinating diversity, a vibrant web of different cultures, languages, traditions and beliefs ( 3 ). Archaeological finds from the Mesolithic and Palaeolithic periods in India and Pakistan represent crucial phases in human prehistory characterized by significant migrations, technological advances, and cultural developments ( 4 - 6 ). The archaeological evidence of various sites, i.e. H. Paleolithic sites such as Soan Valley (Pakistan), Bhimbetka (India), etc., and Mesolithic sites such as Bagor (India), Mehrgarh (Pakistan), etc., across the region provide valuable insights into the lives and movements of early human populations ( 1 , 7 , 8 ). The region encompassing present-day India and Pakistan is home to some of the oldest and most advanced civilizations in human history, such as the Indus Valley Civilization (IVC - c. 3300-1300 BC), the Vedic Civilization (c. 1500-500 BC) and the Mauryan Empire (ca. 322–185 BC), Gandhara civilization (ca. 1500 BC–500 AD) ( 9 - 14 ). The remarkable diversity of the population in this region is evidence of its rich history, centuries of cultural exchange and coexistence of numerous ethnic groups ( 2 , 15 ) These geographical strata (i.e. India and Pakistan) play an important role in the distribution of human migration, which leads to the exchange of culture and language according to different ethnic populations ( 5 , 16 ). One of the largest, fastest and most recent human migrations was observed during the partition of India in 1947 ( 17 , 18 ). This mass movement of people was not only a population shift but also an ethnic and religious reconfiguration of the region, with millions forced to leave their ancestral homes ( 19 , 20 ). Among the various affected populations, the Sindhis living in Pakistan’s Sindh province were particularly affected ( 21 ). Sindh province utilizes one of the ancient archaeological sites (3rd millennium BC) from the Indus Valley Civilization (IVC), namely Mohenjo-Daro ( 22 ). During Partition, a large number (about a million) of non-Muslim Sindhi migrated to Gujarat, Rajasthan, and Punjab ( 23 , 24 ). They speak the Sindhi language, which belongs to the Indo-European language family. Over time, the language underwent various transitions such as Prakrit, Arabic, Persian, etc. due to the influence of different traders and rulers ( 25 , 26 ). From its ancient roots, such as the Indus Valley Civilization, to its present status, Sindhi has absorbed and integrated influences from various languages and cultures, making it a unique and resilient language ( 27 ). The genetic makeup of the Sindhi population also reflects historical migrations and interactions with neighbouring regions ( 28 ). Genetic studies on the maternal markers of the Sindhi population show a wide range of mtDNA haplogroups, indicating diverse maternal ancestry ( 29 ). Many of the mtDNA haplogroups in Sindhis are ancient and have deep roots in the South Asian subcontinent, such as M, R and U ( 30 ). In contrast, haplogroups such as W and HV indicate genetic links with populations from West and Central Asia ( 29 , 31 ). Genetic markers shared with North Indian (Indo-European) and South Indian (Dravidian) populations indicate a history of gene flow between these groups in Sindhis from South Asia ( 32 ). The Arab conquest of Sindh in the 8th century introduced new maternal lineages into the Sindhi gene pool, reflected in the presence of West Eurasian haplogroups ( 29 , 31 , 33 ). The paternal ancestry of the Sindhi population, like their maternal ancestry, reflects a rich and diverse history marked by ancient migrations, invasions and cultural interactions ( 34 , 35 ). The common haplogroups among Sindhis include R1a, J2, J1, L and H, which are widespread in South Asia and adjacent regions ( 35 - 38 ). R1a in Sindhis indicates the influence of Indo-European migrations into the subcontinent about 3,500 years ago, J2 is associated with agricultural communities in the Fertile Crescent and ancient Mesopotamia. Haplogroup L is widespread in the Indian subcontinent and the Middle East ( 36 , 39 ). This haplogroup is widespread in South Asia and is associated with the Dravidian populations; The presence of J1 haplogroups is typical of the Arabian Peninsula, which is less common but is still notable among Sindhis ( 37 , 39 ). The studies on autosomal STR markers also suggest a connection between Sindhis and other populations of Pakistan and Northwest India ( 38 , 40 ). Other genome-wide studies have also shown a relationship between northwest Indian populations such as Gujjar and Ror, and the Sindhis population, proving that they share genetic ancestry with Indo-Europeans and Middle Eastern populations ( 41 ). These connections likely stem from trade, invasions, and other forms of interaction over millennia. Earlier study pointed out the connection between Northwest India (NWI) and Southwest coast India ( 42 ) through much earlier migrations. However, very limited genetic information available about Sindhis living in other parts of India, particularly in the West coast part of India. In the present study, for the first-time, we report genotype data of the Sindhi settlement from the Thane in Maharashtra in the West coastal region of India. This study will examine the common ancestry, assimilation and the past migration history of the Sindhis to South India. Methods For this project, we followed approved guidelines, applied protocols, and obtained approval from the Institutional Ethics Committee of CSIR-CCMB, Hyderabad, India. Blood samples were collected from 13 healthy Sindhi individuals (all Male) from Thane in Maharashtra, India after written informed consent was obtained and DNA was isolated using the standard phenol-chloroform method and used for subsequent genetic analysis. All Indian Sindhi samples (n=13) were genotyped with the Illumina HumanOmniExpress 24 v2 array kit using the manufacturer’s protocol for a total of 642,824 genome-wide single nucleotide polymorphisms. The dataset was merged with the published DNA dataset of contemporary populations from the HGDP( 43 ) and Genome Asia panel ( 44 ) and quality filtering was applied in Plink 1.9 ( 45 ) to include only autosomal markers on 22 chromosomes with a genotyping call -Rate of > 99% to be retained and minor allele frequency >1% (511668 SNPs). Kinship based filtering was performed by removing individuals with first- and second-degree relatives using the KING-robust function implemented in Plink2 ( 45 , 46 ). After all filtering, the final merged dataset included 866 modern individuals genotyped at 511,828 SNPs. To minimize the effect of background LD in PCA and ADMIXTURE-like analyses ( 47 ), markers were further pruned by selecting SNPs in strong LD (r2 > 0.4, window of 200 SNPs, sliding window of 25 SNPs each) using Plink 1.9 ( 45 ). The LD-trimmed data included 365,621 SNPs on the autosome. For all subsequent analysis except PCA and admixture, the full SNP set of 511,828 sites was used. Principal component analysis (PCA) was performed on the merged dataset of modern Eurasia using the Smartpca package implemented in EIGENSOFT 7.2.1 ( 48 ) with default settings. The first two components were recorded to infer genetic variability. The model-based clustering algorithm ADMIXTURE ( 47 ) was executed to infer ancestral genomic components in the Indian Sindhi population. Cross-validation was performed 25 times for 11 ancestral clusters (K=2 to K=12) (Fig. S1). The lowest CV error parameter was obtained at K = 6 and used for downstream analysis. The qp3Pop implementation of the ADMIXTOOLS package ( 49 ) was used to calculate the outgroup F3 statistics. To infer the gene flow of modern Eurasians in the Indian Sindhi population, the F3 statistic was used in the form F3 (Mbuti; SND, X), where X is any modern West Eurasian or South Asian population. (SND = Indian Sindhi). The haplotype-based approach implemented in CHROMOPAINTER ( 50 ) and FineStructure ( 50 ) was used to infer a fine-scale co-ancestry matrix and population clustering, respectively. The data were first phased with SHAPEIT5 ( 51 ) using default parameters, followed by a CHROMOPAINTER run to derive the co-ancestry matrix, first by performing a 10-expectation maximization iteration (EM) with 5 randomly selected chromosomes with a subset of individuals to derive global mutation rate (µ) and switch rate (Ne) parameters. The main algorithm was then run on 22 chromosomes from all individuals to derive the co-ancestry matrix. This matrix was used by FineStructure to infer clustering using a probabilistic model by applying the Markov Chain Monte Carlo (MCMC) method and then deriving a hierarchical tree by merging all clusters with the least change in posterior probability. The run used 500,000 burn-in iterations and 5,000,000 subsequent iterations and the results of each 10,000-iteration saved. Estimates of admixture date and best admixture models were derived with fastGlobeTrotter ( 52 ) using Chromopainter chunklength files. For identity by descent (IBD) analysis, we performed haplotype inference or phasing using three independent runs of Beagle-5.4 ( 53 ). IBD segments were determined from phase data from all three runs separately using refined IBD ( 54 ), then segments from all three runs were combined, and then combined segments were merged using the Merge IBD Segments tool. The IBD release matrix was then recorded using a custom script in R ( 55 ). To derive the best-fitting demographic model and model parameters, we used the parameter optimization method implemented in Moments ( 56 ). For Indian Sindhi, we used a preliminary model based on hypothesis driven from either FastGlobeTrotter ( 52 ) admixture models of Indian Sindhi groups, with much earlier Dai-like admixture or alternatively a more recent pulse in India. For model construction, we used the Python package Demes ( 57 ). Parameter files were created based on the respective Demes models. Two alternative models were used to compare the demographic scenario of Indian Sindhi groups (Supplementary Fig S4 & S5). The site frequency spectrum was calculated from empirical data in VCF format as well as from Demes model specifications using moments. Model parameter optimizations were performed with 500 iterations and the lbfgsb method. Confidence intervals for derived parameters were calculated using the moments . Demes . Inference . uncerts function of Moments ( 56 ). Results Genetic structure in Indian Sindhi The PCA biplots presented includes Indian Sindhi (black), Indo-Europeans (blue), Dravidians (red), Austroasiatic speakers (khaki), Tibeto-Burman (orange), Pakistani groups (forest-green), and North West Indians (light-green) ( Fig 1b ). Interestingly, the Indian Sindhi (black dots) clustered near one extreme of the South Asian cline (but away from main cline) with most of the Pakistan and Northwest Indian (NWI) groups with highest ANI ancestry. Their clustering pattern was more shifted towards the Burusho from Pakistan. This shifting is towards the PCA axis occupied by East/Southeast Asian groups ( Fig 1b ). Two of the Indian Sindhi individuals are shifted towards Indian Indo-Europeans and one of them clustered along with individuals from Konkani population. Download figure Open in new tab Fig 1: Sampling location and population structure of Indian Sindhi. A . Location in the Indian state of Maharashtra (Thane) from West coast, B . PCA biplot of Indian Sindhi (SND) with modern Eurasians, C . Admixture barplot with modern Eurasians (red elipse shows East Asian Khaki colour component). D . Outgroup F3 statistics with modern Eurasians (showing highest allele sharing of SND with NWI) (Dark green colour in NWI). In the unsupervised model-based clustering with ADMIXTURE ( 47 ) using K=6 ( Fig. 1C ), Indian Sindhi formed a unique East Asian component (Khaki) different from most of the Pakistan/NWI groups but this component was also observed in Burusho and only few individuals of Pathan. This ancestral component is maximised among East Asians, Southeast Asians, Indian Tibeto-Burmans and Indian Austroasiatic groups. Besides this both Indian Sindhi and Burusho have typical South Asian Indo-European ancestral components (Blue, Forest green and Orange) ( Fig. 1C ). Allele sharing between Indian Sindhi and modern Eurasians In the outgroup F3 statistics, the Indian Sindhi showed highest allele sharing with the Khatri population (F3 = 0.2648; z = 133.8107) from NWI followed by Kalash (F3 = 0.02639; z = 129.6765) from Pakistan and Gujjar (F3 = 0.02621; z = 133.5288) again from NWI (Supplementary_Table S1). Some of the other top hits were mostly Indian Indo-European populations like Rabadi, Rajput and Brahmin_UP. Fine scale population structure and IBD sharing The fineSTRUCTURE ( 50 ) tree kept all the populations in two major clades, with one major clade included East/Southeast Asians, Andamanese, Indian Tibeto-Burman, Indian Austroasiatic and Dravidians, while other clade incorporated Europe/MidEast, Pakistan/NWI and Indian Indo-Europeans ( Fig 2a ). In this second clade Indian Sindhi forms an altogether separate minor branching with Konkani group from west coast India. Most of the Pakistani Sindhi individual were sharing clades with Pakistan/NWI populations ( Fig 2a ). Download figure Open in new tab Fig 2: Haplotype and IBD sharing statistics. A . fineStructure MCMC tree for Indian Sindhi (SND) with all modern Eurasians, B . Inter-population IBD sharing matrix adjusted for population size, C . Intra-population IBD sharing within group (LOD > 10). In the cross population IBD sharing matrix adjusted for population size, Indian Sindhi showed highest IBD sharing with Konkani population from West coast (Maharashtra) India, followed by Khatri population from Northwest India (NWI) ( Fig 2b ). Whereas, in the intra-population IBD sharing the length distribution is smaller in comparison to most of the modern Eurasians and almost comparable to Pakistani Sindhi ( Fig 2c ). In the within population IBD sharing, we excluded the values below LOD score of 10 in all cases. Admixture modelling and dating The best fit sources for Indian Sindhi in the estimation of the best fitted model and date of admixture using fastGlobeTrotter ( 52 ) were Khatri from Northwest India (NWI) and Dhurwa (an Austronesian proxy) ( Fig 3b ). The best fit date of admixture was approximately 46 GBP (Supplementary Table S2). Download figure Open in new tab Fig 3: Admixture modelling and dates of Indian Sindhi A . ALDER LD decay curve for Burusho and SND (Indian Sindhi), B . fastGlobeTrotter co-ancestry curve fitting for SND (Indian Sindhi), C . Best fit demographic model with admixture date from Dai in SND (Indian Sindhi). Demographic history and demographic parameter estimation We proposed two alternate demographic models for model competition for the Indian Sindhi population, with first model based on our admixture modelling results of Dai admixture, which is evident in model-based Admixture as well as fastGlobeTrotter ( 52 ) modelling and another model with possible admixture with Indian (possibly Austroasiatic) groups as source of Dai-like component. For replicating the admixture history of Indian Sindhi in the prior model, we used Demes ( 57 ). We used Moments’( 56 ) inference optimization function to arrive at best likelihood model and parameters. Of the tested two alternate models, Model1 hypothesize the pulse of admixture from Dai-like source much earlier and with similar event to Burusho in Pakistan (probably through Mongolian invasion), while Model2 hypothesize putative admixture between ANI-ASI and later admixture event with Dai-like source recently from Indian Austroasiatic groups to form the Indian Sindhi group. We selected Model1 (Log-likelihood: -198268.97404874387) ( Fig 3c ) over Model2 (Log-likelihood: -206243.77405798397) based on their likelihood scores, which corroborated well with admixture model inferred from fastGlobeTrotter ( 52 ) run. This recent admixture with Dai was dated to approximately 37.4 GBP (95% C.I. 29.003-45.8005), which is almost comparable to the date estimate in Burusho population from fastGlobeTrotter run and upper limit of 95% C.I. corresponds to Indian Sindhi fastGlobeTrotter estimate (46 GBP) (Supplementary Table 1b). The parameter estimates from best fitted model of Indian Sindhi suggest that there was not significant reduction in effective population size in this group (NA=5200; NF=3260), with noticeable migration rates between ANI and Indian Sindhi (M_French_GroupA =0.00183) (Fig. S3). This effective population size change was more prominent in case of ASI (Paniya as a proxy; Ne=61000 and NeF=854) ( Fig 3c ). Discussion The high level of population diversity and stratification in India is due to multiple waves of migration from outside into the region over millennia and eventual mixing and cultural assimilation. Genetic evidence of later migrations and admixture events is well documented, particularly on the west coast of India, such as among Indian Parsees (Chaubey, Ayub et al. 2017), Cochin Jews (Chaubey, Singh et al. 2016), and Roman Catholic Jews (Kumar, Farias et al. 2021) and migration and local assimilation of warriors clans from Northwest India to Southwest coast ( 42 ). In the present study, we have carried out, for the first time, a detailed investigation of the genetic architecture of a small, isolated and socio-culturally unique Indian Sindhi settlement from the Thane in Maharashtra. Our analysis suggests that This Indian Sindhi settlement in Thane represent a unique group, distinct from the local population in India as well as the Pakistani Sindhi population. Their genetic structure surprisingly indicates their closer affinity to the Burusho-like population from Pakistan, due to the presence of an additional East/Southeast Asian genetic component. This Indian Sindhi subgroup did not form a close group with the Pakistani Sindhi, reflecting their marked contrast with that group. Allele sharing statistics (outgroup F3) using Mbuti as the outgroup and various modern Eurasians as the test group indicate greater shared-drift of Indian Sindhi with populations from Northwest India (NWI) and Pakistan. This reflects their long-standing shared affinity with Pakistan/NWI, as does Pakistani Sindhi. This pattern of genetic affinity of populations from Pakistan (Sindhi, Pathan etc.,) with Northwest Indian is well pointed out through earlier genetic studies ( 38 , 40 , 41 ). Indian Sindhi have a well-documented migration history from Pakistan/NWI, which correlates well with our genetic findings. Of note, the genetic architecture of these Indian Sindhis from Thane in context of local population also equally draw attention, as they share haplotypes with the local Konkani groups from West coast India. This kind of haplotype sharing often reflects very recent gene flow patterns and this is also evident in the IBD chunk sharing. They share larger IBD segments with Konkani population apart from comparatively short segments with Khatri population from Northwest India (NWI). Former represents much recent gene flow while later represents earlier shared genetic history of Indian Sindhi with NWI populations. Pakistani Sindhi also share their genetic history with Pakistan/NWI populations based on earlier study on Y-STR (Anwar et al. 2019; Perveen et al. 2017). Further Northwest Indian populations like Ror and Gujjar showed major affinity with Pakistan and Northwest Indian populations ( 41 ). Therefore, the PCA-based clustering of Indian Sindhi among Pakistani populations as well as haplotype sharing with Khatri largely reflects their long-term shared genetic history with both Pakistani and northwest Indian populations. At this point, it is important to discuss the unique East Asian (Dai-like) component that clearly distinguishes Indian Sindhi settlement in Thane from Pakistani Sindhi. Our haplotype-based admixture modelling suggests that this minor component (∼10%) was introduced at 46.46 GBP with the best-matched surrogates as Khatri (NWI) and Dhurwa (Austroasiatic group). The latter group represents a proxy for an Austronesian (Dai-like) surrogate. Estimates of admixture date using the linkage disequilibrium-based method were also similar (55.47+-19.46). Furthermore, these date estimates were well supported in our demographic modelling, with the admixture timing found to be 37.4 GBP from the Dai-like source group in the best-fit model. Second model which suggest more recent gene flow from a group carrying Dai-like component after Indian Sindhi migration to India, is excluded. Thus, successful model indicates much earlier admixture of this component and the time frame overlaps with the Mongolian (Genghis khan) invasion in Pakistan although other possible source cannot be excluded. Apart from Indian Sindhi and Burusho, Pathan also shows the minor presence of this additional East Asian component. The Late Bronze Age Steppe populations and many Iron age migrations (Saka, Hun, Kushan etc) were having additional East Asian genetic components ( 58 ). Most of the Populations from Pakistan (Balochi, Pathan, Burusho, Sindhi and Hazara) are in geographical proximity and in a transition zone in relation to pre-Historical and Historical migrations. Hence there is possibility of acquiring such component during any of these migration waves, which require much detailed genetic investigations. Although the ALDER-based estimate was comparatively at the high end, the fastGlobeTrotter date estimate was at the upper limit of the 95% confidence interval of the demographic model-based admixture date estimate (95 % C.I. 19-45 GBP). This may be due to other admixture events or noise in ALDER admixture date estimates. Furthermore, parameter estimates in our Indian Sindhi demographic modelling revealed no evidence of a significant founding event or population bottleneck in this group, and there was no significant change in the effective population size. In conclusion, this study presents the first insightful genetic evidence for the ancient origin and unique genetic architecture of Sindhi population settlement from the Thane in Maharashtra in West coast India. The study is the first to report evidence of East Asian admixture in this group much earlier in history than their migration to the West coast of India. Their genetic assimilation with the local majority population (Konkani) reflects their predominantly exogamous nature, which is well reflected in lower IBD exchange within the population. Given the limited sample size of the Indian Sindhi from West coast India in the present study, future efforts with a much larger sample size incorporating Sindhi populations from most of the India along with uniparental and whole genome analysis will reveal more interesting aspects of their population history. Furthermore, further population genetic studies of this kind on different groups will shed light on the heterogeneity of many similar genetic migrations to India. Ethical Approval Informed written consent was obtained from each participant. The project was carried out in accordance with the guidelines approved by the Institutional Ethical Committees of Centre for Cellular and Molecular Biology, Hyderabad, India. All the procedure has been followed according to the recommendations of the Helsinki Declaration. Declaration of interest The authors declare no competing interests. Informed consent Informed written consent was obtained from all the participants involved in the study. Data availability Statement The data supporting the findings of this study are available upon request. Author contributions KT conceptualised the study and recruited the study samples. LK and KT devised the methodology. LK and SN genotyped the genome wide SNP markers. LK performed the data analyses. KT and LK wrote the first draft of the manuscript. KT finalised the report. KT provided feedback on the report. All authors contributed to and have approved the final manuscript. Acknowledgements We thank all the study participants, who volunteered in this study. KT was supported by J C Bose Fellowship (JCB/2019/000027) from the Science and Engineering Research Board (SERB), Department of Science and Technology, Government of India. References 1. ↵ Chakrabarty DK . India: An Archaeological History: Palaeolithic beginnings to early historic foundations : Oxford University Press ; 2009 . 2. ↵ Reich D , Thangaraj K , Patterson N , Price AL , Singh L. Reconstructing Indian population history . Nature . 2009 ; 461 ( 7263 ): 489 – 94 . OpenUrl CrossRef PubMed Web of Science 3. ↵ Khan FD . Preserving the heritage: a case study of handicrafts of Sindh (Pakistan) . 2011 . 4. ↵ Jacobson J. Recent developments in South Asian prehistory and protohistory . Annual Review of Anthropology . 1979 : 467 – 502 . 5. ↵ James HA , Petraglia M. Modern human origins and the evolution of behavior in the later Pleistocene record of South Asia . Current anthropology . 2005 ; 46 ( S5 ): S3 - S27 . OpenUrl CrossRef Web of Science 6. ↵ Blinkhorn J , Petraglia MD . Environments and cultural change in the Indian subcontinent: implications for the dispersal of Homo sapiens in the Late Pleistocene . Current Anthropology . 2017 ; 58 ( S17 ): S463 – S79 . OpenUrl CrossRef 7. ↵ Allchin B , Allchin R. The rise of civilization in India and Pakistan : Cambridge University Press ; 1982 . 8. ↵ Kennedy KA . Prehistoric skeletal record of man in South Asia . Annual review of anthropology . 1980 : 391 – 432 . 9. ↵ Ahmad N , Rehman AU . The emergence of Gandhara Civilization: A politico-historical discourse . Journal of Humanities, Social and Management Sciences (JHSMS) . 2021 ; 2 ( 2 ): 42 – 54 . OpenUrl 10. Ahmed M. Ancient Pakistan-an Archaeological History: Volume III : Harappan Civilization-the Material Culture: Amazon ; 2014 . 11. Dutt RC . A History of Civilization in Ancient India: Vedic and epic ages : Thacker, Spink and Company ; 1889 . 12. Gupta GS . India: From Indus Valley Civlization to Mauryas : Concept Publishing Company ; 1999 . 13. Rose D , Allen R. Ancient civilizations of the world: Scientific e-Resources ; 2018 . 14. ↵ Thapar R. The Mauryan empire in early India . Historical Research . 2006 ; 79 ( 205 ): 287 – 305 . OpenUrl CrossRef 15. ↵ Moorjani P , Thangaraj K , Patterson N , Lipson M , Loh PR , Govindaraj P , et al. Genetic evidence for recent population mixture in India . Am J Hum Genet . 2013 ; 93 ( 3 ): 422 – 38 . OpenUrl CrossRef PubMed 16. ↵ Gadgil M , Joshi N , Manoharan S , Patil S , Prasad US . Peopling of India . The Indian human heritage . 1998 : 100 – 29 . 17. ↵ Zamindar VF-Y. The long partition and the making of modern South Asia: Refugees, boundaries, histories : Columbia University Press ; 2007 . 18. ↵ Bharadwaj P , Khwaja A , Mian A. The big march: migratory flows after the partition of India . Economic and Political Weekly . 2008 : 39 – 49 . 19. ↵ Amrith SS . Migration and diaspora in modern Asia : Cambridge University Press ; 2011 . 20. ↵ Leaning J , Bhadada S. The 1947 partition of British India: Forced migration and its reverberations : SAGE Publishing India ; 2022 . 21. ↵ Kumar P , Kothari R. Sindh, 1947 and beyond . Taylor & Francis ; 2016 . p. 773 – 89 . 22. ↵ Parpola A. The roots of Hinduism: the early Aryans and the Indus civilization : Oxford University Press, USA ; 2015 . 23. ↵ Kothari R. Unbordered Memories: Sindhi Stories of Partition: Penguin Random House India Private Limited ; 2018 . 24. ↵ Boivin M , Lalchandani T. Everyday Religiosity among the Hindu Sindhis of India: Sindhi Identity and the Religious Market in the Era of Social Networks . 2024 : 153 – 72 . 25. ↵ Wadhwani Y. The origin of the Sindhi language . Bulletin of the Deccan College Research Institute . 1981 : 192 – 201 . 26. ↵ Rahman T. Language and politics in a Pakistan province: The Sindhi language movement . Asian Survey . 1995 ; 35 ( 11 ): 1005 – 16 . OpenUrl FREE Full Text 27. ↵ Iyengar AV , Ndhlovu F , Schneider C. Sindhī multiscriptality past and present: A sociolinguistic investigation into community acceptance . 2017 . 28. ↵ Ahmad M , Sinha A , Ghosh S , Kumar V , Davila S , Yajnik CS , et al. Inclusion of Populationspecific Reference Panel from India to the 1000 Genomes Phase 3 Panel Improves Imputation Accuracy . Scientific Reports . 2017 ; 7 ( 1 ). 29. ↵ Bhatti S , Aslamkhan M , Attimonelli M , Abbas S , Aydin HH . Mitochondrial DNA variation in the Sindh population of Pakistan . Australian Journal of Forensic Sciences . 2017 ; 49 ( 2 ): 201 – 16 . OpenUrl CrossRef 30. ↵ Quintana-Murci L , Chaix R , Wells RS , Behar DM , Sayar H , Scozzari R , et al. Where west meets east: the complex mtDNA landscape of the southwest and Central Asian corridor . The American Journal of Human Genetics . 2004 ; 74 ( 5 ): 827 – 45 . OpenUrl CrossRef PubMed Web of Science 31. ↵ Yasmin M , Rakha A , Noreen S , Salahuddin Z. Mitochondrial control region diversity in Sindhi ethnic group of Pakistan . Legal Medicine . 2017 ; 26 : 11 – 3 . OpenUrl CrossRef 32. ↵ Singh M , Sarkar A , Kumar D , Nandineni MR . The genetic affinities of Gujjar and Ladakhi populations of India . Sci Rep . 2020 ; 10 ( 1 ): 2055 . OpenUrl CrossRef PubMed 33. ↵ Bhatti S , Abbas S , Aslamkhan M , Attimonelli M , Trinidad MS , Aydin HH , et al. Genetic perspective of uniparental mitochondrial DNA landscape on the Punjabi population, Pakistan . Mitochondrial DNA Part A . 2018 ; 29 ( 5 ): 714 – 26 . OpenUrl CrossRef 34. ↵ Levi SC . The Indian diaspora in Central Asia and its trade, 1550--1900 . 2001 . 35. ↵ Ikram MS , Mehmood T , Rakha A , Akhtar S , Khan MIM , Al-Qahtani WS , et al. Genetic diversity and forensic application of Y-filer STRs in four major ethnic groups of Pakistan . BMC Genomics . 2022 ; 23 ( 1 ). 36. ↵ Adnan A , Rakha A , Nazir S , Khan MF , Hadi S , Xuan J. Evaluation of 13 rapidly mutating Y-STRs in endogamous Punjabi and Sindhi ethnic groups from Pakistan . International Journal of Legal Medicine . 2019 ; 133 : 799 – 802 . OpenUrl CrossRef PubMed 37. ↵ Mehdi S , Qamar R , Ayub Q , Khaliq S , Mansoor A , Ismail M , et al. The origins of Pakistani populations: evidence from Y chromosome markers . Genomic diversity: applications in human population genetics : Springer ; 1999 . p. 83 – 90 . 38. ↵ Perveen R , Shahid AA , Shafique M , Shahzad M , Husnain T. Genetic variations of 15 autosomal and 17 Y-STR markers in Sindhi population of Pakistan . International journal of legal medicine . 2017 ; 131 : 1239 – 40 . OpenUrl CrossRef PubMed 39. ↵ Qamar R , Ayub Q , Mohyuddin A , Helgason A , Mazhar K , Mansoor A , et al. Y-chromosomal DNA variation in Pakistan . Am J Hum Genet . 2002 ; 70 ( 5 ): 1107 – 24 . OpenUrl CrossRef PubMed Web of Science 40. ↵ Anwar I , Hussain S , Rehman AU , Hussain M. Genetic variation among the major Pakistani populations based on 15 autosomal STR markers . International journal of legal medicine . 2019 ; 133 : 1037 – 8 . OpenUrl CrossRef PubMed 41. ↵ Pathak AK , Kadian A , Kushniarevich A , Montinaro F , Mondal M , Ongaro L , et al. The Genetic Ancestry of Modern Indus Valley Populations from Northwest India . Am J Hum Genet . 2018 ; 103 ( 6 ): 918 – 29 . OpenUrl CrossRef PubMed 42. ↵ Kumar L , Chowdhari A , Sequeira JJ , Mustak MS , Banerjee M , Thangaraj K. Genetic Affinities and Adaptation of the South-West Coast Populations of India . Genome Biol Evol . 2023 ; 15 ( 12 ): evad225 . OpenUrl CrossRef PubMed 43. ↵ Bergstrom A , McCarthy SA , Hui RY , Almarri MA , Ayub Q , Danecek P , et al. Insights into human genetic variation and population history from 929 diverse genomes . Science . 2020 ; 367 ( 6484 ): 1339 -+. OpenUrl CrossRef 44. ↵ GenomeAsia KC . The GenomeAsia 100K Project enables genetic discoveries across Asia . Nature . 2019 ; 576 ( 7785 ): 106 – 11 . OpenUrl CrossRef PubMed 45. ↵ Chang CC , Chow CC , Tellier LC , Vattikuti S , Purcell SM , Lee JJ . Second-generation PLINK: rising to the challenge of larger and richer datasets . Gigascience . 2015 ; 4 ( 2047-217X (Electronic)):7. 46. ↵ Manichaikul A , Mychaleckyj JC , Rich SS , Daly K , Sale M , Chen WM . Robust relationship inference in genome-wide association studies . Bioinformatics . 2010 ; 26 ( 22 ): 2867 – 73 . OpenUrl CrossRef PubMed Web of Science 47. ↵ Alexander DH , Novembre J , Lange K. Fast model-based estimation of ancestry in unrelated individuals . Genome Res . 2009 ; 19 ( 9 ): 1655 – 64 . OpenUrl Abstract / FREE Full Text 48. ↵ Patterson N , Price AL , Reich D. Population structure and eigenanalysis . PLoS Genet . 2006 ; 2 ( 12 ): e190 . OpenUrl CrossRef PubMed 49. ↵ Patterson N , Moorjani P , Luo Y , Mallick S , Rohland N , Zhan Y , et al. Ancient admixture in human history . Genetics . 2012 ; 192 ( 3 ): 1065 – 93 . OpenUrl Abstract / FREE Full Text 50. ↵ Lawson DJ , Hellenthal G , Myers S , Falush D. Inference of population structure using dense haplotype data . PLoS Genet . 2012 ; 8 ( 1 ): e1002453 . OpenUrl CrossRef PubMed 51. ↵ Hofmeister RJ , Ribeiro DM , Rubinacci S , Delaneau O. Accurate rare variant phasing of whole-genome and whole-exome sequencing data in the UK Biobank . Nat Genet . 2023 ; 55 ( 7 ): 1243 – 9 . OpenUrl CrossRef PubMed 52. ↵ Wangkumhang P , Greenfield M , Hellenthal G. An efficient method to identify, date, and describe admixture events using haplotype information . Genome Res . 2022 ; 32 ( 8 ): 1553 – 64 . OpenUrl Abstract / FREE Full Text 53. ↵ Browning BL , Zhou Y , Browning SR . A One-Penny Imputed Genome from Next-Generation Reference Panels . Am J Hum Genet . 2018 ; 103 ( 3 ): 338 – 48 . OpenUrl CrossRef PubMed 54. ↵ Browning BL , Browning SR . Improving the accuracy and efficiency of identity-by-descent detection in population data . Genetics . 2013 ; 194 ( 2 ): 459 – 71 . OpenUrl Abstract / FREE Full Text 55. ↵ Team RC . R: A language and environment for statistical computing . R Foundation for Statistical Computing , Vienna, Austria 2021 . 56. ↵ Jouganous J , Long W , Ragsdale AP , Gravel S. Inferring the Joint Demographic History of Multiple Populations: Beyond the Diffusion Approximation . Genetics . 2017 ; 206 ( 3 ): 1549 – 67 . OpenUrl Abstract / FREE Full Text 57. ↵ Gower G , Ragsdale AP , Bisschop G , Gutenkunst RN , Hartfield M , Noskova E , et al. Demes: a standard format for demographic models . Genetics . 2022 ; 222 ( 3 ): iyac131 . OpenUrl CrossRef PubMed 58. ↵ Narasimhan VM , Patterson N , Moorjani P , Rohland N , Bernardos R , Mallick S , et al. The formation of human populations in South and Central Asia . Science . 2019 ; 365 ( 6457 ): eaat7487 . OpenUrl Abstract / FREE Full Text View the discussion thread. Back to top Previous Next Posted March 03, 2025. Download PDF Supplementary Material Email Thank you for your interest in spreading the word about bioRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following Investigating the demographic history of Sindhi population inhabited in West coast India Message Subject (Your Name) has forwarded a page to you from bioRxiv Message Body (Your Name) thought you would like to see this page from the bioRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share Investigating the demographic history of Sindhi population inhabited in West coast India Lomous Kumar , Suraj Nongmaithem , Sachin Kumar , Kumarasamy Thangaraj bioRxiv 2025.03.01.640946; doi: https://doi.org/10.1101/2025.03.01.640946 Share This Article: Copy Citation Tools Investigating the demographic history of Sindhi population inhabited in West coast India Lomous Kumar , Suraj Nongmaithem , Sachin Kumar , Kumarasamy Thangaraj bioRxiv 2025.03.01.640946; doi: https://doi.org/10.1101/2025.03.01.640946 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Genetics Subject Areas All Articles Animal Behavior and Cognition (7633) Biochemistry (17681) Bioengineering (13890) Bioinformatics (41929) Biophysics (21446) Cancer Biology (18586) Cell Biology (25492) Clinical Trials (138) Developmental Biology (13374) Ecology (19897) Epidemiology (2067) Evolutionary Biology (24308) Genetics (15606) Genomics (22497) Immunology (17736) Microbiology (40385) Molecular Biology (17175) Neuroscience (88584) Paleontology (666) Pathology (2831) Pharmacology and Toxicology (4822) Physiology (7641) Plant Biology (15149) Scientific Communication and Education (2045) Synthetic Biology (4293) Systems Biology (9822) Zoology (2271)
Text is read by the "Ask this paper" AI Q&A widget below.
Extraction quality varies by source — PMC NXML preserves structure
cleanly, OA-HTML may include some navigation residue, and OA-PDF can
have broken hyphenation. The publisher copy
(via DOI)
is the canonical version.