Full text
27,344 characters
· extracted from
preprint-html
· click to expand
SAI: A Python Package for Statistics for Adaptive Introgression | bioRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-M677548'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search New Results SAI: A Python Package for Statistics for Adaptive Introgression View ORCID Profile Xin Huang , View ORCID Profile Simon Chen , View ORCID Profile Josef Hackl , View ORCID Profile Martin Kuhlwilm doi: https://doi.org/10.1101/2025.04.19.649497 Xin Huang 1 Department of Evolutionary Anthropology, University of Vienna , Vienna, Austria 2 Human Evolution and Archaeological Sciences (HEAS), University of Vienna , Vienna, Austria Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Xin Huang For correspondence: xin.huang{at}univie.ac.at Simon Chen 1 Department of Evolutionary Anthropology, University of Vienna , Vienna, Austria Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Simon Chen Josef Hackl 1 Department of Evolutionary Anthropology, University of Vienna , Vienna, Austria Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Josef Hackl Martin Kuhlwilm 1 Department of Evolutionary Anthropology, University of Vienna , Vienna, Austria 2 Human Evolution and Archaeological Sciences (HEAS), University of Vienna , Vienna, Austria Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Martin Kuhlwilm Abstract Full Text Info/History Metrics Preview PDF Abstract Adaptive introgression is an important evolutionary process, yet widely used summary statistics—such as the number of uniquely shared sites and the quantile of the derived allele frequencies in such sites—lack accessible implementations, limiting reproducibility and methodological clarity. Here, we present SAI, a Python package for computing these statistics, and apply it to three datasets. First, using the 1000 Genomes Project data, we replicated previously reported candidate regions and identified additional ones, including a region detected by studies using supervised deep learning. Second, reanalysis of a Lithuanian genome dataset revealed no candidates in the HLA region. Finally, we investigated bonobo introgression into central chimpanzees and identified a candidate region that overlaps a high-frequency Denisovan-introgressed haplotype block reported in modern Papuans—an intriguing co-occurrence across divergent lineages. Discrepancies with prior results highlight the importance of transparent and reproducible analysis workflows, especially as machine learning becomes increasingly prevalent in evolutionary genomics. Adaptive introgression refers to the transfer of genetic material between distantly related lineages through gene flow, increasing the fitness of the recipient lineage. To detect candidate genomic regions of adaptive introgression, some studies primarily utilized ad-hoc approaches—first applying methods to detect natural selection and introgression separately, then identifying their intersection as candidate regions ( Fijarczyk and Babik 2015 ; Nye et al. 2018 ; Zhang X et al. 2023). Furthermore, supervised machine learning methods, such as genomatnn, MaLAdapt, and ERICA, have been introduced specifically for detecting loci of adaptive introgression ( Gower et al. 2021 ; Zhang X et al. 2023; Zhang Y et al. 2023). Meanwhile, dedicated summary statistics for this purpose have been proposed by Racimo et al. (2017) (see Supplementary Table S1), such as the number of uniquely shared sites ( U statistic) and the quantile of the derived allele frequencies in such sites ( Q statistic). These statistics remain invaluable, particularly in cases where the demographic model is unknown ( Gower et al. 2021 ; Romieu et al. 2024 ). Despite their utility, these statistics lack a readily available implementation, requiring researchers to develop custom scripts for their analyses, which could harm accessibility and reproducibility of research. To address these gaps, we developed SAI (version 1.0.1), a Python package designed for calculating the U and Q statistics, providing a robust framework for future studies on adaptive introgression (Supplementary Material). We excluded the R D statistic proposed by Racimo et al. (2017) , as its implementation may require excluding zero-denominator cases and introduce systematic bias (Supplementary Material). We first applied SAI to the 1000 Genomes Project Phase 3 dataset ( The 1000 Genomes Project Consortium 2015 ), as previously reported in Racimo et al. (2017) (see Supplementary Material). Here, the Altai Neanderthal and Denisovan genomes were used as source or donor populations, that is, the populations presumed to have contributed introgressed material ( Huang et al. 2022 ). Candidate regions were identified based on the overlap of outliers from both the U 50 and Q 95 statistics, computed using derived allele frequencies (Supplementary Table S2). These regions were then annotated using ANNOVAR (version 2022Oct05), with annotations from RefSeq, dbNSFP v4.2c, and dbSNP build 150 ( Sherry et al. 2001 ; Wang et al. 2010 ; O’Leary et al. 2016 ; Liu et al. 2020 ). We successfully replicated several previously reported candidate regions, including those in BNC2 , a gene associated with human pigmentation and may subject to natural selection in modern Europeans ( Cheng et al. 2021 ; Huang et al. 2021 ). We also identified additional candidate regions, including one on chromosome 20 that was not reported by Racimo et al. (2017) but was later detected using supervised deep learning ( Gower et al. 2021 ). However, we did not recover the region chr17:18880001–18920000 (hg19 coordinates), which Racimo et al. (2017) identified as a candidate region of Denisovan-specific introgression. We observed discrepancies in U 50 and Q 95 values between our results and those reported by Racimo et al. (2017) within the same genomic regions (Supplementary Table S2). To validate our SAI results, we compared allele counts computed by our implementation with those obtained using BCFtools (version 1.21; Danecek et al. 2021 ) and found them to be consistent (Supplementary Material). We also noted that the original definition of the U statistic is not specifically based on the derived allele (Supplementary Table S1). Therefore, we extended our implementation to compute these statistics using either allele 0 or allele 1, depending on which allele meets the required conditions (Supplementary Material). Despite this adjustment, our results (Supplementary Table S3) still did not align with those of Racimo et al. (2017) . We further filtered out low-quality variants in the archaic hominin genomes (Supplementary Material), but the discrepancies persisted (Supplementary Tables S4 and S5). Although the original definition of the Q statistic is based on the derived allele, Racimo et al. (2017) did not provide explicit details on how ancestral states were inferred when analyzing real data. A similar issue arises in MaLAdapt (Zhang X et al. 2023), which uses the U and Q statistics as input features for its machine learning algorithm (Extra-Trees classifiers). While MaLAdapt explicitly defines these statistics as based on the derived allele, the source of ancestral allele information is not specified. Since Racimo et al. (2017) did not make their code or detailed data processing steps publicly available, the precise causes of these discrepancies remain unresolved. We then applied SAI to the Lithuanian genomes published by Urnikyte et al. (2023) to screen for candidates of adaptive introgression from Neanderthals (Supplementary Material). As in our previous study ( Hackl and Huang 2025 ), we used BEAGLE 5.4 (version 06Aug24.a91) to impute missing genotypes, using all European populations from the 1000 Genomes Project as the reference panel ( Browning et al. 2018 ; Browning et al. 2021 ). We did not identify any variants in the HLA region as candidates for adaptive introgression (Supplementary Tables S6 and S7), consistent with our earlier findings using MaLAdapt ( Hackl and Huang 2025 ). Notably, the region chr20:62110001–62220000 (hg19 coordinates) was detected as a Neanderthal-introgressed candidate, in agreement with both our results above (Supplementary Tables S2–S5) and a previous study that reported this region in European populations from the 1000 Genomes Project ( Gower et al. 2021 ). However, because our approach requires the allele to reach high frequency in the target population, we may miss candidates under weak or ongoing selection—particularly those maintained at low or intermediate frequencies. Finally, we explored bonobo introgression into central chimpanzees using a curated great ape genome dataset (Han et al. 2025) with SAI, with western chimpanzees as non-introgressed reference population (Supplementary Material). Compared to human genome data, the great ape genomes showed higher values of the U statistic ( Figure 1 ), indicating a substantial number of shared variants between central chimpanzees and bonobos—consistent with a previous study ( de Manuel et al. 2016 ). Among our candidate regions (Supplementary Tables S8 and S9), we found that three genes— ADGRL4, GALNTL6 , and SORCS1 —had been previously reported as candidates for positive selection within bonobo-introgressed regions ( Nye et al. 2018 ). Interestingly, we also identified the region chr14:58400001– 58450000 (hg38 coordinates), which contains TOMM20L, TIMM9 , and KIAA0586 , as a candidate region (Supplementary Table S9). A previous study reported that this region assigned to high-frequency Denisovan-introgressed haplotype blocks in modern Papuans ( Jacobs et al. 2019 ). Download figure Open in new tab Figure 1. Joint distributions of genome-wide U and Q values computed using derived allele frequencies. (A) Joint distribution of U 50 and Q 95 values from the 1000 Genomes Project. Following Racimo et al. (2017) , the filtering criteria required allele frequency 0.5 in the EUR population, and fixed in both the Neanderthal and Denisovan genomes. (B) Joint distribution of U 90 and Q 95 values from the Lithuanian genomes. The criteria required allele frequency 0.9 in the Lithuanian population, and fixed in the Neanderthal genome. (C) Joint distribution of U 90 and Q 95 values from genomes of the Pan clade. The criteria required allele frequency 0.9 in central chimpanzees, and fixed in bonobos. Our study highlights the importance of terminological precision, such as whether allele frequencies are calculated using ancestral or derived alleles, and technical transparency, such as how ancestral alleles are inferred. Ambiguous definitions can lead to inconsistent interpretations and even propagation of errors in subsequent research. Likewise, the absence of publicly available implementations hinders reproducibility and independent validation. When methodological descriptions are unclear—for example, in the handling of ancestral alleles without explicit documentation—a well-documented implementation, such as a workflow built with Snakemake ( Mölder et al. 2021 ), becomes an essential complement, clarifying the intended computational procedures. As interest grows in applying advanced computational methods, particularly machine learning and deep learning ( Schrider and Kern 2018 ; Christin et al. 2019 ; Andermann 2020 ; Morimoto et al. 2021 ; Lürig et al. 2021 ; Borowiec et al. 2022 ; Callier 2022 ; Pichler and Hartig 2023 ; Korfmann et al. 2023 ; Yelman and Jay 2023; Huang et al. 2024 ; Mo et al. 2024 ; Wu 2024 ), to evolutionary biology, it is essential to ensure both rigorous theoretical formulations and accessible, reproducible implementations to uphold the reliability and integrity of scientific research ( Huang 2024 ). Competing interests The authors declare no conflict of interests. Author contributions X.H. designed the study. X.H. implemented SAI. X.H., S.C., J.H., and M.K. analyzed the data and wrote the manuscript. Data availability The 1000 Genomes Project Phase 3 dataset can be downloaded from, https://ftp.ncbi.nih.gov/1000genomes/ftp/release/20130502/ , last accessed April 19, 2025. The Lithuanian genomes can be downloaded from https://data.mendeley.com/datasets/d2xt5hdm5j/2 , last accessed April 19, 2025. The Altai Neanderthal genome can be downloaded from http://cdna.eva.mpg.de/neandertal/Vindija/VCF/Altai/ , last accessed April 19, 2025. The Denisovan genome can be downloaded from http://cdna.eva.mpg.de/neandertal/Vindija/VCF/Denisova/ , last accessed April 19, 2025. BED files specifying variants that passed quality filters in the Altai Neanderthal genome can be downloaded from http://cdna.eva.mpg.de/neandertal/Vindija/FilterBed/Altai/ , last accessed April 19, 2025. BED files specifying variants that passed quality filters in the Denisovan genome can be downloaded from http://cdna.eva.mpg.de/neandertal/Vindija/FilterBed/Denisova/ , last accessed April 19, 2025. The bonobo and chimpanzee genomes can be downloaded from https://phaidra.univie.ac.at/pfsa/o_2066302/merged_segregating/Pan/Pan_wild_filtered/ , last accessed April 19, 2025. The ancestral sequences based on the hg19 reference genome can be found in https://ftp.ensembl.org/pub/release-75/fasta/ancestral_alleles/homo_sapiens_ancestor_GRCh37_e71.tar.bz2 , last accessed April 19, 2025. The FASTA file for the reference genome hg38 can be found in https://hgdownload.soe.ucsc.edu/goldenPath/hg38/bigZips/hg38.fa.gz , last accessed April 19, 2025. The FASTA file for the reference genome rheMac10 can be found in https://hgdownload.soe.ucsc.edu/goldenPath/rheMac10/bigZips/rheMac10.fa.gz , last accessed April 19, 2025. The file for liftover chain from hg38 to rheMac10 can be found in https://hgdownload.soe.ucsc.edu/goldenPath/hg38/liftOver/hg38ToRheMac10.over.chain.gz , last accessed April 19, 2025. The source code of SAI can be found in https://github/xin-huang/sai , last accessed April 19, 2025. The Snakemake workflow for reproducing the analysis can be found in https://github.com/xin-huang/sai-analysis , last accessed April 19, 2025. Acknowledgements We thank Florian Schmidt and Serhat Inceer for testing the SAI package. We also acknowledge the Life Science Compute Cluster at the University of Vienna for providing computing resources. This project has been funded by the Vienna Science and Technology Fund (WWTF) [10.47379/VRG20001] to M.K. Funding Vienna Science and Technology Fund, , VRG20001 References ↵ Andermann T. 2020 . Advancing evolutionary biology: Genomics, Bayesian statistics, and machine learning . https://gupea.ub.gu.se/handle/2077/66848 , last accessed April 19, 2025 . ↵ Borowiec ML , Dikow RB , Frandsen PB , McKeeken A , Valentini G , White AE . 2022 . Deep learning as a tool for ecology and evolution . Methods Ecol Evol 13 : 1640 – 1660 . OpenUrl CrossRef ↵ Browning BL , Zhou Y , Browning SR . 2018 . A one-penny imputed genome from next generation reference panels . Am J Hum Genet 103 : 338 – 348 . OpenUrl CrossRef PubMed ↵ Browning BL , Tian X , Zhou Y , Browning SR . 2021 . Fast two-stage phasing of large-scale sequence data . Am J Hum Genet 108 : 1880 – 1890 . OpenUrl CrossRef PubMed ↵ Callier V. 2022 . Machine learning in evolutionary studies comes of age . Proc Natl Acad Sci USA 119 : e2205058119 . OpenUrl CrossRef PubMed ↵ Cheng JY , Stern AJ , Racimo F , Nielsen R. 2021 . Detecting selection in multiple populations by modeling ancestral admixture components . Mol Biol Evol 39 : msab294 . OpenUrl ↵ Christin S , Hervet É , Lecomte N. 2019 . Applications for deep learning in ecology . Methods Ecol Evol 10 : 1632 – 1644 . OpenUrl CrossRef ↵ Danecek P , Bonfield JK , Liddle J , Marshall J , Ohan V , Pollard MO , Whitwham A , Keane T , McCarthy SA , Davies RM , et al. 2021 . Twelve years of SAMtools and BCFtools . GigaScience 10 : giab008 . OpenUrl CrossRef PubMed ↵ de Manuel M , Kuhlwilm M , Frandsen P , Sousa VC , Desai T , Prado-Martinez J , Hernandez-Rodriguez J , Dupanloup I , Lao O , Hallast P , et al. 2016 . Chimpanzee genomic diversity reveals ancient admixture with bonobos . Science 354 : 477 – 481 . OpenUrl Abstract / FREE Full Text ↵ Fijarczyk A , Babik W. 2015 . Detecting balancing selection in genomes: Limits and prospects . Mol Ecol 14 : 3529 – 3545 . OpenUrl ↵ Gower G , Picazo PI , Fumagalli M , Racimo F. 2021 . Detecting adaptive introgression in human evolution using convolutional neural networks . eLife 10 : e64669 . OpenUrl CrossRef PubMed ↵ Hackl J , Huang X. 2025 . Revisiting adaptive introgression at the HLA genes in Lithuanian genomes with machine learning . Infect Genet Evol 127 : 105708 . OpenUrl CrossRef PubMed Han S , Riyahi, Huang X , Kuhlwilm. 2025 . A curated great ape genome diversity panel . bioRixv . https://doi.org/10.1101/2025.02.18.638799 , last accessed April 19, 2025 . ↵ Huang X , Kruisz P , Kuhlwilm M. 2022 . sstar: A Python package for detecting archaic introgression from population genetic data with S* . Mol Biol Evol 39 : msac212 . OpenUrl CrossRef PubMed ↵ Huang X. 2024 . Developing machine learning applications for population genetic inference: Ensuring precise terminology and robust implementation . EcoEvoRixv . doi: 10.32942/X2N90M , last accessed April 19, 2025 . OpenUrl CrossRef ↵ Huang X , Rymbekova A , Dolgova O , Lao O , Kuhlwilm M. 2024 . Harnessing deep learning for population genetic inference . Nat Rev Genet 25 : 61 – 78 . OpenUrl CrossRef PubMed ↵ Huang X , Wang S , Jin L , He Y. 2021 . Dissecting dynamics and differences of selective pressures in the evolution of human pigmentation . Biol Open 10 : bio056523 . OpenUrl Abstract / FREE Full Text ↵ Jacobs GS , Hudjashov G , Saag L , Kusuma P , Darusallam CC , Lawson DJ , Mondal M , Pagani L , Ricaut FX , Stoneking M , et al. 2019 . Multiple deeply divergent Denisovan ancestries in Papuans . Cell 177 : 1010 - 1021.e32 . OpenUrl CrossRef PubMed ↵ Korfmann K , Gaggiotti OE , Fumagalli M. 2023 . Deep learning in population genetics . Genome Biol Evol 15 : evad008 . OpenUrl CrossRef PubMed ↵ Liu X , Li C , Mou C , Dong Y , Tu Y. 2020 . dbNSFP v4: A comprehensive database of transcript-specific functional predictions and annotations for human nonsynonymous and splice-site SNVs . Genome Med 12 : 103 . OpenUrl CrossRef PubMed ↵ Lürig MD , Donoughe S , Svensson EI , Porto A , Tsuboi M. 2021 . Computer vision, machine learning, and the promise of phenomics in ecology and evolutionary biology . Front Ecol Evol 9 : 642774 . OpenUrl CrossRef ↵ Mo YK , Hahn MW , Smith ML . 2024 . Applications of machine learning in phylogenetics . Mol Phylo Evol 196 : 108066 . OpenUrl CrossRef PubMed ↵ Mölder F , Jablonski KP , Letcher B , Hall MB , Tomkins-Tinch CH , Sochat V , Forster J , Lee S , Twardziok SO , Kanitz A , et al. 2021 . Sustainable data analysis with Snakemake . F1000 Res 10 : 33 . OpenUrl ↵ Morimoto J , Ponchon A , Sofronov G , Travis J. 2021 . Applications of machine learning to evolutionary ecology data . Front Ecol Evol 9 : 797319 . OpenUrl CrossRef ↵ Nye J , Laayouni H , Kuhlwilm M , Mondal M , Marques-Bonet T , Bertranpetit J. 2018 . Selection in the introgressed regions of the chimpanzee genome . Genome Biol Evol 10 : 1132 – 1138 . OpenUrl CrossRef PubMed ↵ Pichler M , Hartig F. 2023 . Machine learning and deep learning–A review for ecologists . Method Ecol Evol 14 : 994 – 1016 . OpenUrl CrossRef ↵ Racimo F , Marnetto D , Huerta-Sánchez E. 2017 . Signatures of archaic adaptive introgression in presentday human populations . Mol Biol Evol 34 : 296 – 317 . OpenUrl CrossRef PubMed ↵ Romieu J , Camarata G , Crochet PA , de Navascués M , Leblois R , Rousset F. 2024 . Performance evaluation of adaptive introgression classification methods . bioRixv . doi: 10.1101/2024.06.12.598278 , last accessed April 19, 2025 . OpenUrl Abstract / FREE Full Text ↵ O’Leary NA , Wright MW , Brister R , Ciufo S , Haddad D , McVeigh R , Rajput B , Robbertse B , Smith-White B , Ako-Adjei D , et al. 2016 . Reference sequence (RefSeq) database at NCBI: Current status, taxonomic expansion, and functional annotation . Nucleic Acids Res 44 : D733 – D745 . OpenUrl CrossRef PubMed ↵ Sherry ST , Ward MH , Kholodov M , Baker J , Phan L , Smigielski EM , Sirotkin K. 2001 . dbSNP: The NCBI database of genetic variation . Nucleic Acids Res 29 : 308 – 311 . OpenUrl CrossRef PubMed Web of Science ↵ Schrider DR , Kern AD . 2018 . Supervised machine learning for population genetics: A new paradigm . Trends Genet 34 : 301 – 312 . OpenUrl CrossRef PubMed ↵ The 1000 Genomes Project Consortium . 2015 . A global reference for human genetic variation . Nature 526 : 68 – 74 . OpenUrl CrossRef PubMed ↵ Urnikyte A , Masiulyte A , Pranckeniene L , Kučinskas V. 2023 . Disentangling archaic introgression and genomic signatures of selection at human immunity genes . Infect Genet Evol 116 : 105528 . OpenUrl CrossRef PubMed ↵ Wang K , Li M , Hakonarson H. 2010 . ANNOVAR: Functional annotation of genetic variants from next-generation sequencing data . Nucleic Acids Res 38 : e164 . OpenUrl CrossRef PubMed ↵ Wu CI . 2024 . Vision statement: 2025 SMBE meeting in Beijing–Looking forward to the next decade . https://smbe2025.scimeeting.cn/en/web/index/25070_2230628 , last accessed April 19, 2025 . Yelmen B , Jay F. 2023 . An overview of deep generative models in functional and evolutionary genomics . Annu Rev Biomed Data Sci 6 : 173 – 189 . OpenUrl CrossRef PubMed Zhang X , Kim B , Singh A , Sankararaman S , Durvasula A , Lohmueller KE . 2023 . MaLAdapt reveals novel targets of adaptive introgression from Neanderthals and Denisovans in worldwide human populations . Mol Biol Evol 40 : msad001 . OpenUrl CrossRef PubMed Zhang Y , Zhu Q , Shao Y , Jiang Y , Ouyang Y , Zhang L , Zhang W. 2023 . Inferring historical introgression with deep learning . Syst Biol 72 : 1013 – 1038 . OpenUrl CrossRef PubMed View the discussion thread. Back to top Previous Next Posted April 22, 2025. Download PDF Email Thank you for your interest in spreading the word about bioRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following SAI: A Python Package for Statistics for Adaptive Introgression Message Subject (Your Name) has forwarded a page to you from bioRxiv Message Body (Your Name) thought you would like to see this page from the bioRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share SAI: A Python Package for Statistics for Adaptive Introgression Xin Huang , Simon Chen , Josef Hackl , Martin Kuhlwilm bioRxiv 2025.04.19.649497; doi: https://doi.org/10.1101/2025.04.19.649497 Share This Article: Copy Citation Tools SAI: A Python Package for Statistics for Adaptive Introgression Xin Huang , Simon Chen , Josef Hackl , Martin Kuhlwilm bioRxiv 2025.04.19.649497; doi: https://doi.org/10.1101/2025.04.19.649497 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Evolutionary Biology Subject Areas All Articles Animal Behavior and Cognition (7618) Biochemistry (17636) Bioengineering (13860) Bioinformatics (41847) Biophysics (21401) Cancer Biology (18536) Cell Biology (25424) Clinical Trials (138) Developmental Biology (13353) Ecology (19860) Epidemiology (2067) Evolutionary Biology (24287) Genetics (15583) Genomics (22463) Immunology (17701) Microbiology (40300) Molecular Biology (17141) Neuroscience (88434) Paleontology (666) Pathology (2825) Pharmacology and Toxicology (4813) Physiology (7633) Plant Biology (15107) Scientific Communication and Education (2042) Synthetic Biology (4285) Systems Biology (9808) Zoology (2268)
Text is read by the "Ask this paper" AI Q&A widget below.
Extraction quality varies by source — PMC NXML preserves structure
cleanly, OA-HTML may include some navigation residue, and OA-PDF can
have broken hyphenation. The publisher copy
(via DOI)
is the canonical version.