Full text
75,297 characters
· extracted from
preprint-html
· click to expand
Rare splice and missense variants with evidence of pathogenicity in consanguineous families with autosomal recessive intellectual disability from Pakistan | medRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-P4HH5NV'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search Rare splice and missense variants with evidence of pathogenicity in consanguineous families with autosomal recessive intellectual disability from Pakistan Abdul Waheed , Robert Eveleigh , Danielle Perley , Janick St-Cyr , François Lefebvre , Abdul Hameed Khan , Zarqash Majeed , Abrish Majeed , Katerina Trajanoska , Raquel Cuella-Martin , Claude Bhérer , Ghazanfar Ali , Vincent Mooser , Daniel Taliun doi: https://doi.org/10.1101/2024.01.08.23299914 Abdul Waheed 1 Department of Biotechnology, University of Azad Jammu and Kashmir , Muzaffarabad, Pakistan Find this author on Google Scholar Find this author on PubMed Search for this author on this site Robert Eveleigh 2 Department of Human Genetics, Victor Philip Dahdaleh Institute of Genomic Medicine, McGill University , Montreal, Quebec, Canada Find this author on Google Scholar Find this author on PubMed Search for this author on this site Danielle Perley 2 Department of Human Genetics, Victor Philip Dahdaleh Institute of Genomic Medicine, McGill University , Montreal, Quebec, Canada Find this author on Google Scholar Find this author on PubMed Search for this author on this site Janick St-Cyr 2 Department of Human Genetics, Victor Philip Dahdaleh Institute of Genomic Medicine, McGill University , Montreal, Quebec, Canada Find this author on Google Scholar Find this author on PubMed Search for this author on this site François Lefebvre 2 Department of Human Genetics, Victor Philip Dahdaleh Institute of Genomic Medicine, McGill University , Montreal, Quebec, Canada Find this author on Google Scholar Find this author on PubMed Search for this author on this site Abdul Hameed Khan 1 Department of Biotechnology, University of Azad Jammu and Kashmir , Muzaffarabad, Pakistan Find this author on Google Scholar Find this author on PubMed Search for this author on this site Zarqash Majeed 4 Department of Zoology, University of Azad Jammu and Kashmir , Muzaffarabad, Pakistan Find this author on Google Scholar Find this author on PubMed Search for this author on this site Abrish Majeed 5 Bagh College of Arts and Modern Sciences Rera , Bagh, Pakistan Find this author on Google Scholar Find this author on PubMed Search for this author on this site Katerina Trajanoska 2 Department of Human Genetics, Victor Philip Dahdaleh Institute of Genomic Medicine, McGill University , Montreal, Quebec, Canada 3 CERC Program in Genomic Medicine, Victor Philip Dahdaleh Institute of Genomic Medicine, McGill University , Montreal, Quebec, Canada Find this author on Google Scholar Find this author on PubMed Search for this author on this site Raquel Cuella-Martin 2 Department of Human Genetics, Victor Philip Dahdaleh Institute of Genomic Medicine, McGill University , Montreal, Quebec, Canada 3 CERC Program in Genomic Medicine, Victor Philip Dahdaleh Institute of Genomic Medicine, McGill University , Montreal, Quebec, Canada Find this author on Google Scholar Find this author on PubMed Search for this author on this site Claude Bhérer 2 Department of Human Genetics, Victor Philip Dahdaleh Institute of Genomic Medicine, McGill University , Montreal, Quebec, Canada 3 CERC Program in Genomic Medicine, Victor Philip Dahdaleh Institute of Genomic Medicine, McGill University , Montreal, Quebec, Canada Find this author on Google Scholar Find this author on PubMed Search for this author on this site Ghazanfar Ali 1 Department of Biotechnology, University of Azad Jammu and Kashmir , Muzaffarabad, Pakistan Find this author on Google Scholar Find this author on PubMed Search for this author on this site Vincent Mooser 2 Department of Human Genetics, Victor Philip Dahdaleh Institute of Genomic Medicine, McGill University , Montreal, Quebec, Canada 3 CERC Program in Genomic Medicine, Victor Philip Dahdaleh Institute of Genomic Medicine, McGill University , Montreal, Quebec, Canada Find this author on Google Scholar Find this author on PubMed Search for this author on this site Daniel Taliun 2 Department of Human Genetics, Victor Philip Dahdaleh Institute of Genomic Medicine, McGill University , Montreal, Quebec, Canada 3 CERC Program in Genomic Medicine, Victor Philip Dahdaleh Institute of Genomic Medicine, McGill University , Montreal, Quebec, Canada Find this author on Google Scholar Find this author on PubMed Search for this author on this site For correspondence: daniel.taliun{at}mcgill.ca Abstract Full Text Info/History Metrics Preview PDF Abstract Intellectual disability (ID) is a neurodevelopmental disorder affecting up to 1-3% of people worldwide. Genetic factors, including rare de novo or rare homozygous mutations, explain many cases of autosomal dominant or recessive forms of ID. ID is clinically and genetically heterogeneous, with hundreds of genes associated with it. In this study, we performed high-depth whole-genome sequencing of twenty individuals from five consanguineous families from Pakistan, with nine individuals affected by mild or severe ID. We identified one splice and five missense rare variants (at allele frequencies below 0.001%) in a homozygous state in the affected individuals with supporting and moderate evidence of pathogenicity based on guidance from the American College of Medical Genetics and Genomics. These six variants mapped to different genes ( SRD5A3 , RDH11 , RTF2 , PCDHA2 , ADAMTS17 , and TRPC3 ), and only SRD5A3 had previously been known to cause ID. The p.Tyr169Cys mutation inside SRD5A3 was predicted to be deleterious and affect protein structure by multiple in silico tools. In addition, we found one missense mutation, p.Pro1505Ser, inside UNC13B with conflicting evidence of pathogenic and benign effects. Further functional studies are required to confirm the pathogenicity of these variants and understand their role in ID. Our findings provide additional needed information for interpreting rare variants in the genetic testing of ID. 1. INTRODUCTION Studies estimate the prevalence of ID in the general population to be from less than 1% and up to 3%, depending on the country [ 1 – 3 ]. The main characteristics of ID are decreased mental abilities (such as reasoning, recognition, abstract thinking, and judgement) and, consequently, decreased adaptive functioning (such as communication, social participation, and independent living) [ 3 ]. ID belongs to a group of neurodevelopmental disorders that typically manifest in early childhood and can co-occur with other neurodevelopmental disorders (e.g., autism spectrum disorder, ASD) or physical conditions (e.g., cerebral palsy). Depending on the severity level (mild, moderate, severe, profound), ID may be diagnosed within the first two years of life (severe) or may not be identifiable until school age (mild) [ 3 ]. Early and accurate ID diagnosis is essential for initiating early intervention strategies to improve the quality of life of those affected. Both genetic and non-genetic factors can cause ID. Non-genetic factors include events and environmental exposures during and after pregnancy (e.g., maternal disease, exposure to hazardous chemicals) and early childhood (e.g., severe head injury, meningitis, chronic social deprivation). Genetic factors consist of chromosomal aneuploidies and rare mutations, including copy-number variations (CNVs), single-nucleotide variations (SNVs), and structural variations (SVs). Rare de novo mutations may lead to autosomal dominant ID (ADID) and are estimated to explain up to 60% of severe ID cases [ 4 ]. Rare mutations inherited in the homozygous state by individuals, typically in consanguineous families, are associated with autosomal recessive ID (ARID). They are estimated to explain up to 57% of ARID cases in consanguineous families [ 4 ]. ID is a genetically heterogeneous disease with up to 700 genes linked to it or associated disorders by 2015 [ 4 ], and around 1,300 genes listed in the approved gene panel for ID (v5.0 from 22 March 2023) available at Genomics England’s PanelApp [ 5 ]. However, the prevailing view is that more ID-causal genes are yet to be discovered, and the effects of all possible mutations within ID-causing genes have not yet been assessed [ 4 , 6 , 7 ]. Searching and describing novel rare mutations inside known ID genes and identifying new candidate genes may improve the molecular diagnosis of ID and lead to a better understanding of the underlying molecular disease mechanisms. This study focuses on identifying rare mutations responsible for ARID in five consanguineous families from Azad Jammu and Kashmir, Pakistan. We performed WGS in twenty family members (nine affected by mild, moderate, or severe ID) with a projected average depth of coverage of 30X. We called SNVs and short insertions and deletions (indels) and followed them with extensive quality checks, variant filtering, and in silico variant effect predictions. Then, we performed mapping of autosomal recessive variants for ID in each family using only high-quality, rare and predicted potentially damaging variants. In this stage, we limited our analysis to rare variants within protein-coding gene boundaries, which include coding and untranslated regions (UTRs) and introns. We identified seven mutations in homozygous state in affected family members placed within seven genes ( SRD5A3 , RDH11 , RTF2 , PCDHA2 , ADAMTS17 , UNC13B , and TRPC3 ), where only SRD5A3 was established previously as causal of ID. Each of the seven mutations showed one moderate and one supportive evidence of pathogenicity as defined in the American College of Medical Genetics and Genomics (ACMG) guidance for the interpretation of sequence variants. The first section of the results describes the variant filtering and quality control across all sequenced individuals. We then describe the findings in each family separately. The last section describes the results of in silico modelling of protein structures with the identified possibly damaging variants. 2. METHODS 2.1. Study description Five consanguineous families with Autosomal Recessive Intellectual Disability (ARID) were identified in various urban and rural areas of Azad Jammu and Kashmir, Pakistan ( Supplementary Figure 1 ). Only families including two or more persons suffering from concordant phenotypes related to ARID, having parental consanguinity and showing the autosomal recessive mode of inheritance, were enrolled in the study. For simplicity, analyzed families are referred to as ARID-76, ARID-78, ARID-79, ARID-80, and ARID-81. The ORIC (Office of Research Innovation and Commercialization) and DSAR (Director of Advanced Study and Research) boards of the University of Azad Jammu and Kashmir, Muzaffarabad, Pakistan, approved the study ethics protocol. Written informed consent forms were signed by the guardians of each family enrolled in current research for performing genetic analyses and publishing face photographs whenever necessary. The DNA sequencing and genetic data analyses were also approved by the Research Ethics Office (IRB) of the Faculty of Medicine and Health Sciences at McGill University, Montreal, Canada. 2.2. Interviews and clinical evaluation Participating families were interviewed in person. Parents or guardians were interviewed regarding the family history of the affected persons to rule out the involvement of any environmental factors. No environmental factors were identified. A trained physician conducted a physical examination during a home visit to exclude additional morphological, behavioural, neurological, or other organ system abnormalities. Affected individuals were assessed for the presence of ID (mild, moderate, or severe), developmental delay, deficient behaviour, and other co-morbid phenotypes (e.g., communication, behaviour with others, aggressiveness, anxiety, sleeping time, awareness of daily life needs, locomotion, and movements). From three-to five-generation pedigrees were collected and analyzed to infer the mode of inheritance of the disease. Based on the information provided, pedigrees were structured following the approach described in Bennett et al ., 2008 [ 8 ]. Genetic relationships among individuals and disease inheritance patterns prevailing within the families have been elucidated through pedigrees. Each family pedigree was anonymized and assigned a unique identifier consisting of the ARID-prefix and a two-digit number, and the backward mapping is known only within the research group. 2.3. Blood sample collection and DNA extraction Venous blood was collected from affected and unaffected family members. Blood was sampled and preserved in vacutainer tubes containing ethylene diamine tetra acetate (EDTA). Blood samples were transported from the field in the ice box. Samples were stored in the refrigerator at the Department of Biotechnology University of Azad Jammu and Kashmir lab. Genomic DNA was extracted from peripheral white blood cells using the standard phenol-chloroform method. 2.4. Sequencing, alignment, and quality checks Whole genome sequencing (WGS) was performed at the McGill Genome Centre in October 2022, targeting 300 bp length fragments and 30X depth of sequencing coverage (DP). Raw reads sequencing data was aligned to the GRCh38 reference genome, sorted by position, indels re-aligned, marked for duplicates and quality controlled using GenPipes version 4.3.0 [ 9 ]. The aligned reads were stored in BAM files. The average DP of WGS across individuals was 27.82X (SE =1.92X) ( Supplementary Table 1 ). On average, 90.52% (SE=0.26%) of the sequenced base pairs were covered by more than ten high-quality reads (i.e., DP > 10X). There were no differences in average DP and percent of base pairs with DP>10X between affected and unaffected individuals and no individuals with significantly lower average DP. ( Supplementary Figure 5 ). The average Freemix contamination score estimated using VerifyBamId2 [ 10 ] was 3.33×10 -4 (SE = 2.26×10 - 4 ), suggesting no evidence of DNA contamination. The estimated number of copies of sex chromosomes derived from the sequencing data was consistent with the reported sex phenotype ( Supplementary Figure 6 ). Principal component analysis (PCA) of the sequenced data showed that all individuals had Central South Asian genetic ancestry, consistent with the known demographic history of the local population ( Supplementary Figure 7 ). 2.5. Variant calling and quality checks GATK’s HaplotypeCaller v4.3.0 [ 11 ] was used for variant calling. Variant calls were generated in the standard variant calling files (VCFs) – one VCF per individual. Indels were left-aligned using “bcftools norm” v1.13 [ 12 ] to generate normalized VCFs. We used three filters to identify and exclude low-quality variants: GATK CNN1D (convolutional neural network, 1D model), GATK CNN2D (convolutional neural network, 2D model), and GATK Hard Filters with GATK’s recommended thresholds ( Supplementary Figure 2 ). Then, we used consensus filtering to keep only those variants that passed all the abovementioned filters ( Supplementary Figure 3 ). We processed the individual BAM files to compute the depth of coverage at each chromosomal position for each sample. Then, we aggregated the depth of coverage at each chromosomal position across all samples and generated an accessibility mask – a list of regions where all individuals were sequenced with 10X or higher depth of coverage. Using the generated accessibility mask, we subset variants for each individual to create a final per-individual VCF, which includes only variants overlapping regions in the accessibility mask. Finally, we merged individual VCFs into a single VCF. We kept only those variants which passed GATK filters across all individuals. When an individual did not have an alternate allele at a specific position within the accessibility mask, the genotype was set to 0/0. The total number of SNVs and indels is reported in Supplementary Table 2 . 2.6. Mapping of autosomal recessive variants for ID We annotated genetic variants with the Variant Effect Predictor (VEP) tool (release 109) [ 13 ]. To map autosomal recessive variants for ID, we considered only variants which are missense, nonsense (stop gain, stop lost, start lost), splice site (splice donors, splice acceptors), frameshifts, and are within 5’ and 3’ UTRs. We kept only those variants which were predicted as potentially damaging by at least one method: PolyPhen > 0.15, SIFT 20. We considered two alternate allele frequency (AF) thresholds for selecting rare variants based on gnomAD v3.1.2 [ 14 ] and 1000 Genomes Project Phase 3 [ 15 ] datasets (i.e., AF must be less than the selected threshold in both datasets): AF < 0.01% and AF < 0.001% ( Supplementary Figure 4 ). For the selected predicted potentially damaging rare variants, we looked at their states in each ARID family. We kept only those which are homozygous (i.e., genotype 1/1) in affected members of the family, heterozygous (i.e., genotype 1/0) in non-affected parents; and heterozygous or absent (i.e., genotype 0/0) in non-affected siblings. For the variants that passed the filters mentioned above, we looked up whether they were previously reported in the ClinVar database (accessed in November 2023) [ 16 ]. 2.7. Conservation The conservation of the ancestral allele across species was assessed with phyloP and phastCons conservation scores [ 17 – 21 ] computed based on the multiple alignments of 100 vertebrates from the UCSC Genome Browser GRCh38/hg38 release [ 22 ] or GERP conservation scores [ 23 ] calculated based on the multiple alignments of 91 mammals from the Ensembl genome browser release 110 [ 24 ]. The phyloP method assesses conservation at individual nucleotides, ignoring the effects of their neighbours, and calculates -log( P -value) under a null hypothesis of neutral evolution. Positive phyloP scores correspond to conservation. The phastCons method computes the probability that a nucleotide belongs to a conserved element, i.e., it considers the neighbouring nucleotides. Similarly to phyloP, the GERP method measures conservation at individual nucleotides and compares the number of observed substitutions to the expected under no functional constraint. Positive GERP scores represent conservation. Our analyses classified the ancestral allele as conserved if at least one of the following thresholds was satisfied: phyloP > 1.9, phastCons > 0.5, GERP > 2.7 [ 25 ]. Additionally, we visually inspected the presence of the ancestral allele across multiple alignments of ten primates in the Ensembl genome browser release 110. 2.8. Gene constraints We used the following gene constraint metrics available from gnomAD v2.1.1 and v4.0.0 [ 26 ]: the ratio of the observed to expected (o/e) number of loss-of-function (LoF) variants in a gene – values close to 0 indicate that the gene is intolerant to LoF mutations (the recommended threshold for the LoF intolerant gene is when the 90% CI’s upper bound of o/e LoF [LOEUF] is less than 0.35 in gnomAD v2.1.1 and less than 0.6 in gnomAD v4.0.0); the o/e number of missense variants in a gene – values close to 0 indicate that gene is intolerant to missense mutations; missense Z-score – high scores indicate that gene is intolerant to missense mutations (the recommended threshold for the missense intolerant gene is when Z-score > 3.09 [ 27 ]). We also used the sub-genic missense constraint score [ 28 ], denoted γ, available from the DECIPHER online tool [ 29 ]. The γ score varies between 0 and 1, where smaller values indicate fewer observed missense variants than expected. 2.9. Prediction of protein stability We applied four methods to evaluate the effects of the mutations on protein structures: FoldX[ 30 ], Rosetta[ 31 , 32 ], Venus[ 33 ], and Missense3D[ 34 ]. We looked for the experimentally derived protein structures (e.g., using cryo-electron microscopy (cryo-EM) or X-ray crystallography) from the Protein Data Bank in Europe (PDBe) [ 35 ] and for structural predictions from the AlphaFold Protein Structure Database (AlphaFold DB) [ 36 ]. The FoldX, Rosetta, and Venus methods predict the change in Gibbs free energy, denoted as ΔΔG and measured in kcal/mol units, between wild-type and mutant protein structures. Mutations resulting in ΔΔG < -1 are considered to stabilize the protein, -1 < ΔΔG 1 - to destabilize the protein [ 37 ]. We ran FoldX v5 locally using default settings: 1) we minimized the original protein structures with the RepairPDB command before estimating ΔΔG; 2) for each mutation, we estimated ΔΔG using the BuildModel command with five runs and calculated the mean ΔΔG value. We ran Rosetta r351 locally following the Cartesian ΔΔG protocol [ 38 ]: 1) we used FastRelax[ 39 – 41 ] with the ref2015_cart weights[ 38 ] to minimize the original protein structure; 2) for each mutation, we estimated ΔΔG using the cartesian_ddg application [ 37 ] with options described in Frenz et al. [ 38 ] and five iterations; 3) we applied the 1/2.94 scaling factor to the average ΔΔG [ 37 ]. We used the MichelANGlo web application [ 42 ] to run Venus, which uses PyRosetta[ 43 ] and the speed-optimized protocol based on the REF15 score function [ 44 ] and provides scaled ΔΔG estimates. We ran Missense3D through the Missense3D web application [ 34 ] with default settings and by uploading the protein structures downloaded from PDBe (if available) or AlphaFoldDB. The Missense3D pipeline evaluates the impact of a mutation on 17 features related to protein structure and highlights features most likely altered by the mutation. 3. RESULTS 3.1. Variant discovery, quality control and filtering We identified 4,031,141 (SE=13,231) SNVs and 928,353 (SE=8,962) indels on average per individual ( Supplementary Table 2 ). We didn’t observe systematic differences in the number of identified variants between affected and unaffected individuals ( Supplementary Figure 8 ). After applying variant-level filtering, the average number of SNVs and indels per individual were 3,712,456 (SE=13,013) and 880,956 (SE=7,965), respectively. In total, we identified 8,325,148 high-quality, unique genetic variants across all individuals placed within genomic regions where each individual had more than ten high-quality reads (i.e., DP>10X) ( Supplementary Table 3 ). We further limited our search to 155,658 non-synonymous variants inside protein-coding regions, splicing regions, and 5’ and 3’ UTRs. Of these, 11,341 variants were predicted as potentially damaging by at least one of the three bioinformatic tools: PolyPhen, SIFT, and CADD. Among these predicted potentially damaging variants, 3,049 had AF < 0.01% in the gnomAD v3.1.2 and 1000 Genomes Project Phase 3 variant databases, with 1,734 remaining when applying a more stringent AF < 0.001% threshold. 3.2. The p.Tyr169Cys mutation inside the known SRD5A3 gene in the ARID-76 family ARID-76 is a five-generation pedigree in which consanguineous parents have three affected and two unaffected children ( Figure 1 ). WGS was done on three individuals: IV-1 (unaffected parent), V-1 (unaffected sibling) and V-4 (affected sibling). The affected sibling (V-4) shows moderate intellectual impairment with night blindness. They express themselves through non-verbal communication and have limited facial recognition skills. They require help with self-care activities. The second affected sibling (not sequenced) is more severely affected, completely non-verbal, with impaired facial recognition capabilities. Download figure Open in new tab Figure 1. Anonymized partial family trees. Symbols represent family members. Parallel dual lines indicate consanguineous marriages. Shaded and empty symbols represent affected and unaffected family members, respectively. Slanted crossed lines represent deceased family members. Every generation within a pedigree is referred to by Roman numerals (I, II, III, IV) from top to bottom, while the positions of family members are depicted using Arabic numbers (1, 2, 3, 4). WGS was performed on labeled individuals: ARID-76 family – IV-1, V-1 and V-4; ARID-78 family – III-1, III-2, IV-1, and IV-2; ARID-79 family – II-1 and III-1; ARID-80 family – III-1, III-2, and IV-1; ARID-81 family – IV-1 and V-1. We found a single predicted potentially damaging variant in the homozygous state in the sequenced affected child but not in the sequenced unaffected relatives ( Table 1 ). The affected individual carries two copies of the alternate allele G of the missense variant rs754837473 A>G (p.Tyr169Cys) in the SRD5A3 gene. In contrast, the unaffected parent has only one copy, and the unaffected sibling has no copies of this allele. The alternate allele G was previously reported by two submitters in the ClinVar database (accession ID: VCV002506565.2) as having uncertain significance [ 45 ]. Only one submitter provided a description, although very brief, of a patient condition: the alternated allele G was found in a homozygous state in a young (< 9-year-old) female of South-East Asian ancestry with SRD5A3 -congenital disorder of glycosylation (SRD5A3-CDG, MIM: 612379), showing a global developmental delay with nystagmus and characterized as self-absorbed, having a happy demeanour, and hand-flapping. In our analyses, the variant was predicted to be potentially damaging by all three in silico methods, Polyphen, SIFT, and CADD ( Table 2 ). The comparative genomic analyses suggested that the ancestral allele A is conserved (PhyloP = 9.1, PhastCons = 1.0), and it was present across ten primates ( Supplementary Table 4 ). Only a single copy of the alternate allele G was found in the gnomAD v3.1.2 non-neuro database (AF=6.6×10 -6 ), which excludes cases with neurological or psychiatric diseases, and it was reported in an individual of European genetic ancestry. Twelve copies of allele G were found in the gnomAD v4.0.0 database (AF=7.4×10 -6 ), which includes 416,555 exomes from the UK Biobank [ 46 ], and all observed copies were in the heterozygous state. Furthermore, 8 of 12 reported carriers were of South Asian genetic ancestry, suggesting a higher prevalence of allele G in this population. Although, in general, the SRD5A3 gene tolerates LoF and missense mutations ( Supplementary Table 5 ), nonsense and frameshift mutations in homozygous and compound heterozygous states in this gene have been implicated in two genetic disorders related to ID. One is SRD5A3-CDG (MIM: 612379), which may result in ID, ophthalmologic and cerebellar defects, and ichthyosiform skin lesions [ 47 – 49 ]. The other disorder is Kahrizi syndrome (KHRZ) (MIM: 612713), which has overlapping features with SRD5A3-CDG – it is an autosomal recessive neurodevelopmental disorder characterized by ID, cataracts, coloboma, kyphosis, and coarse facial features [ 50 ]. View this table: View inline View popup Download powerpoint Table 1. Mapped autosomal recessive variants for ID. * - affected individual. FID – family ID. IID – individual ID. Genotypes: 0 – reference allele, 1 – alternate allele. DP – depth of sequencing at variant position. View this table: View inline View popup Download powerpoint Table 2. Population frequencies, conservation scores and deleteriousness scores for variants with homozygous alternate genotypes in affected individuals but not in unaffected relatives. Values in bold satisfy the defined thresholds: PhyloP > 1.9, PhastCons > 0.5, GERP > 2.9, PolyPhen > 0.15, SIFT 20. AF – alternate allele frequency. N/A – not applicable. To summarize, applying the ACMG guidelines for the variants interpretation [ 51 ], we found one moderate and one supportive evidence of pathogenicity. The first evidence, moderate (ACMG PM2), is that the variant is absent from controls or at extremely low frequency if recessive. The second evidence, supportive (ACMG PP3), is that there are multiple lines of computational evidence supporting a deleterious effect. Thus, the variant can be classified as uncertain significance. However, given that the missense variant rs754837473 A>G was already reported in a homozygous state in a carrier with overlapping phenotype, we propose rs754837473 A>G as the most likely cause of ID in the ARID-76 family. 3.3. Splice and missense variants inside the RDH11 and RTF2 genes in the ARID-78 family ARID-78 is a four-generation pedigree in which all children of consanguineous parents are affected by ID ( Figure 1 ). WGS was performed on four individuals: III-1, III-2 (unaffected, parents), IV-1 (affected sibling) and IV-2 (affected sibling). Both children show mild intellectual impairment with slightly aggressive behaviour. We found two predicted potentially damaging variants in homozygous states in both affected siblings but not in their parents ( Table 1 ). The first one was a splice donor variant chr14:67692437 C>T (c.349+1C>T) inside the RDH11 gene, where both affected individuals had two copies of alternate allele T, while their parents had only one copy. The CADD method predicted the damaging effect of this variant ( Table 2 ), while PolyPhen and SIFT methods were inapplicable to this non-coding variant. The SpliceAI method [ 52 ] predicted a splicing donor loss in the canonical transcript with high confidence (delta score = 0.99) due to alternate allele T. The comparative genomic analyses suggested the conservation of the ancestral allele C (PhyloP = 7.0, PhastCons = 1.0, GERP = 2.2), which was present across ten primates ( Supplementary Table 6 ). The variant was not present in ClinVar or gnomAD databases. The number of missense and LoF variants inside the RDH11 gene reported in gnomAD v4.0.0 is slightly lower than expected, but the difference is not large enough to conclude that the RDH11 gene may not tolerate missense or LoF mutations ( Supplementary Table 5 ). Several nonsense mutations in the compound heterozygous state in the RDH11 gene were previously reported as pathogenic in a single family of Italian-American descent, where three siblings had retinal dystrophy, juvenile cataracts, and short stature syndrome (RDJCSS), experiencing psychomotor delays from early childhood, including learning difficulties (MIM: 616108) [ 53 ]. The second was a missense variant rs530872363 G>A (p.Asp25Asn) inside the RTF2 gene. Like the first variant, both affected siblings had two copies of the alternate allele A, while both parents had only one copy. The variant was predicted to be ‘damaging’ by Polyphen, SIFT, and CADD ( Table 2 ). The comparative genomic analyses suggested that the ancestral allele G is conserved across vertebrates (PhyloP = 9.0, PhastCons = 1.0, GERP = 3.7), and it was present across ten primates ( Supplementary Table 7 ). There were no entries for rs530872363 G>A in the ClinVar database. The gnomAD v4.0.0 database had 97 individuals (AF= 6.0×10 -5 ) carrying a single copy of the alternate allele A and no homozygous carriers. 88 out of 97 carriers in the gnomAD v4.0.0 database were of South Asian genetic ancestry (AF=1.0×10 -3 ), indicating the higher prevalence of this allele in this population. Only seven heterozygous carriers were reported in gnomAD v3.1.2 (no homozygous carriers), and all seven carriers were also present in the gnomAD v3.1.2 non-neuro database. The numbers of missense and LoF mutations observed in the RTF2 gene in the gnomAD v2.1.1 and v4.0.0 databases are lower than expected. Still, this is only suggestive that this gene may be intolerant to both classes of mutations ( Supplementary Table 5 ). This gene was not known previously for ARID or any other rare condition with a similar phenotype. However, RTF2 protein levels in the brain were associated with Alzheimer’s disease in recent proteome-wide association studies (PWASs) with mixed evidence of causality [ 54 , 55 ]. In summary, both chr14:67692437 C>T and rs530872363 G>A have one moderate (ACMG PM2) and one supporting (ACMG PP3) evidence of being pathogenic. Although both variants show evidence of co-segregation with disease in multiple affected family members, none of the genes are known to cause the disease. Thus, the ACMG PP1 criteria for supporting evidence of pathogenicity may not apply. The genotyping of the third affected sibling may help further prioritize one variant over another if one is observed in the heterozygous state. However, lacking additional functional evidence, any remaining variant will still be classified as of uncertain significance. 3.4. Two missense variants inside PCDHA2 and ADAMTS17 in the ARID-79 family ARID-79 is a three-generation pedigree where consanguineous parents have two affected and three unaffected children ( Figure 1 ). WGS was done on two family members: the parent (II-1) and the affected child (III-1). The affected child (III-1) has a severe form of ID. They recognize their parents but display aggressive behaviour towards new people. Similarly, their affected sibling (not sequenced) shows profound intellectual impairment with aggressive behaviour. Both affected siblings depend on the family for assistance with daily activities. We found two predicted potentially damaging variants in the homozygous state in the affected child but not in the parent ( Table 1 ). The first was the missense variant rs376199248 C>A (p.Asn451Lys) in the PCDHA2 gene. The affected child had two alternate allele A copies, while the unaffected parent had one copy. Only Polyphen and SIFT predicted the variant to be damaging, while the CADD score was just below our threshold ( Table 2 ). Although the ancestral allele C was present across ten primates ( Supplementary Table 8 ), none of the applied algorithms suggested a conservation of it (phyloP=-0.423, phastCons=0.037, GERP=-1.51). The variant was not previously reported in the ClinVar database. The gnomAD v4.0.0 database reported 16 heterozygous alternate allele A carriers (AF=9.91×10 -6 ), 13 of non-Finnish European ancestry, and no homozygous. The gene-level constraint scores indicate that the PCDHA2 gene tolerates missense and LoF mutations ( Supplementary Table 5 ). The PCDHA2 has not been previously implicated in ARID development. However, it is a part of the PCDHA gene cluster, which encodes a family of cadherin-like cell surface proteins expressed in neurons and present at synaptic junctions. Missense mutations in the PCDHA3 gene were previously reported in patients with restless leg syndrome (RLS), which affects the nervous system and muscles [ 56 ]. Moreover, multiple CNVs were reported in patients with developmental delay (DD)/ID and ASD in a region containing the PCDHA gene cluster [ 57 , 58 ] The second was a missense variant chr15:100331016 G>C (p.Ile163Met) mapped to ADAMTS17 . The affected child had two alternate allele C copies, while the parent carried only one copy. The Polyphen, SIFT, and CADD methods predicted this variant to be damaging. The ancestral allele G was predicted to belong to a conserved region (phastCons = 0.994) and was present across ten primates ( Supplementary Table 9 ). Still, the remaining methods (GERP and PhyloP) did not indicate conservation at the level of individual nucleotides. The variant was not reported in the ClinVar database, and only two alleles in heterozygous carriers of South Asian and non-Finnish European genetic ancestries were reported in gnomAD v4.0.0. The gene constraint metrics reported in the gnomAD databases indicate that ADAMTS17 tolerates missense and LoF mutations (only gnomAD v2.1.1 reported a lower number of LoF mutations than expected, but this difference was insufficient to claim for LoF intolerance) ( Supplementary Table 5 ). Mutations inside ADAMTS17 were not reported for ARID or related phenotypes. Splice-site, frameshift, and truncating homozygous mutations were found in the ADAMTS17 gene in Indian [ 59 ] and Saudi Arabian [ 60 ] families with the Weill-Marchesani syndrome 4 (WMS 4) (MIM: 613195), a connective tissue disorder. ADAMTS17 was highly expressed in the adult lung, brain, eye, and retina in Saudi Arabian families. It’s been reported that 10-17% of the WMS cases may display mild ID [ 61 ]. Similar to the ARID-78 family, both identified rare variants in ARID-79 have one moderate (ACMG PM2) and one supporting (ACMG PP3) evidence of pathogenicity and can be classified as of uncertain significance. Although the genotyping of the second affected and the other three unaffected siblings may help eliminate one of the variants, it will not allow us to claim pathogenicity with certainty. 3.5. The p. Pro1505Ser mutation inside the UNC13B gene in the ARID-80 family ARID-80 is a four-generation pedigree in which consanguineous parents have two affected and five unaffected children ( Figure 1 ). WGS was performed on both unaffected parents (III-1 and III-2) and one affected child (IV-1). The affected child (IV-1) shows severe intellectual impairment and aggressive behaviour toward others. They need assistance with daily living activities. The other affected sibling (not sequenced) started showing mild intellectual impairment in their 20s, whereas the proband started displaying intellectual impairment during their first several years. We didn’t find any predicted potentially damaging variants with population-level AF below 0.001% in this family. However, we found one predicted potentially damaging variant after relaxing our AF threshold to 0.01%. The sequenced affected child had two copies of the alternate allele at the missense variant rs142697917 C>T (p.Pro1505Ser) inside the UNC13B gene, while both parents had only one copy ( Table 1 ). The Polyphen, SIFT, and CADD methods predicted this variant to be damaging. All three algorithms, phyloP, phastCons, and GERP, suggested that ancestral allele C is conserved across vertebrates and present across ten primates ( Supplementary Table 10 ). The variant was not reported in ClinVar. The gnomAD v4.0.0 database reported 778 carriers of the alternate allele (AF=4.8×10 -4 ), including four homozygous carriers. Most carriers, 509 (including all homozygote carriers), were of South Asian genetic ancestry (AF=5.59×10 -3 ), indicating the higher prevalence of this variant in this population. The one homozygote carrier was present in the gnomAD v3.1.2 database of controls. The gnomAD databases reported depletion of LoF and missense mutations in UNC13B relative to the expected numbers, which may indicate some degree of gene intolerance towards these mutations. ( Supplementary Table 5 ). In general, mutations within the UNC13B gene have not been previously associated with ARID. However, UNC13B is known to be expressed in the brain (predominantly in the cerebral cortex), and rare nonsense, splice, and missense mutations (in compound heterozygous and heterozygous states) in the gene were reported in Chinese families with partial epilepsy [ 62 ]. Multiple CNVs were reported in patients with DD/ID and ASD in a region containing the UNC13B gene [ 57 , 58 ]. Taken together, although the rs142697917 C>T has one supporting evidence of being pathogenic based on computational evidence (ACMG PP3) and may still be viewed as of extremely low frequency in other databases (i.e. satisfy the moderate evidence of pathogenicity ACMG PM2), it also has one strong evidence of being benign. It is observed in healthy adult individuals in the homozygous state (ACMG BS2), while we expect full penetrance at an early age. Despite this, rs142697917 C>T can still be classified as of uncertain significance. Further genotyping of other siblings may be needed to resolve the conflicting evidence of a benign effect with certainty. 3.6. The p.Ile793Phe mutation inside the TRPC3 gene in the ARID-81 family ARID-81 is a five-generation pedigree in which consanguineous parents have four affected (one deceased) and one unaffected child ( Figure 1 ). WGS was performed on two family members: the unaffected parent (IV-1) and the affected child (V-1). The affected child (V-1) shows severe intellectual impairment, is non-verbal, and entirely depends on their parents for essential self-care activities. We found only one predicted potentially damaging variant in the homozygous state in the sequenced affected child but not in the parent. The affected child had two copies of the alternate allele A at the missense variant rs145430158 T>A (p.Ile793Phe) inside the TRPC3 gene, while the parent had only one copy ( Table 1 ). Only SIFT predicted the damaging effect of this variant ( Table 2 ). In contrast, the PolyPhen method predicted benign effects, and the CADD score for this variant was 11, below our set threshold of 20. The phastCons method estimated that the ancestral allele T belongs to the conserved element with a probability of 0.863, and the ancestral allele was present across ten primates ( Supplementary Table 11 ). Other computational methods assessing individual nucleotides didn’t suggest conservation. The variant was not reported in ClinVar. The gnomAD v4.0.0 database reported 148 heterozygote carriers (no homozygous) of the alternate allele (AF= 9.2×10 -5 ): 53 heterozygotes were of non-Finish European genetic ancestry (AF = 4.6×10 -5 ), and 85 were of South Asian genetic ancestry (AF = 9.3×10 -4 ). The gene-level constraint metrics showed suggestive evidence that the TRPC3 gene is intolerant to LoF mutations and strong evidence that it is intolerant to missense mutations. The region-level constraint metrics also strongly supported intolerance to missense mutations ( Supplementary Table 5 ). The TRPC3 gene is expressed in the human cerebellum, and the genetic mouse model with the heterozygous gain-of-function mutation in TRPC3 shows motor and coordination defects associated with the progressive loss of Purkinje cells in the cerebellum [ 63 ]. Thus, several studies hypothesized the causal role of this gene in cerebellar ataxia in humans. However, the systematic screening of 108 patients with suspected genetic forms of ataxias failed to reveal any mutations that could significantly contribute to this condition within TRPC3 [ 64 ]. Alternatively, Fogel et al ., 2014 [ 65 ] suggested that the heterozygous missense mutation in a 40-year-old man of European ancestry in the TRPC3 gene may cause adult-onset spinocerebellar ataxia-41 (SCA41; MIM: 616410), but additional evidence for more robust conclusions were needed. To summarize, like other discussed mutations, the rs145430158 T>A mutation can be classified as of uncertain significance: it has moderate evidence of pathogenicity because of its low frequency (ACMG PM2); it has supporting evidence of pathogenicity from in silico tools (i.e. region-level conservation and deleteriousness as defined by the SIFT score; ACMG PP3). Because the TRPC3 gene is highly expressed in the cerebellum and may play a role in spinocerebellar ataxia (a neurogenerative disorder), it is possible to speculate that the described mutation may cause ID in the ARID-81 family. The genotyping of other siblings may help resolve this speculation. 3.7. The in silico modelling predicts changes in SRD5A3 protein structure caused by the p.Tyr169Cys mutation We performed in silico protein structure modelling for the six reported missense mutations (splice mutation in RDH11 was excluded from this analysis due to limitations of the applied protein modelling tools) ( Table 3 ). The missense mutation rs754837473 A>G, identified in the ARID-76 family, causes the Tyr169Cys substitution in SRD5A3 protein-coding transcripts. All four protein structure modelling methods (Venus, FoldX, Rosetta, and Missense3D) predicted the destabilizing effects of the Tyr169Cys mutation on the protein corresponding to the canonical transcript. The Venus, FoldX, and Rosetta methods predicted positive changes in thermodynamic stability, ΔΔG > 2 kcal/mol, destabilizing the protein, while Missense3D indicated the cavity alteration in the protein structure. All methods used the in silico AlphaFold predictions as starting structures. The rest of the mutations either didn’t show significant effects on protein structure based on the applied in silico methods, or the available in silico protein models were not highly accurate. For example, the pLDDT accuracy scores for the available in silico protein models for the UNC13B gene were below 90, which makes significant positive changes in thermodynamic stability predicted by the FoldX and Rosetta unreliable. View this table: View inline View popup Download powerpoint Table 3. In silico predicted mutation effects on protein folding and structure. Only protein-coding transcripts are reported. The transcript IDs in bold correspond to the Ensembl canonical transcripts. The UniProt IDs in bold correspond to the canonical isoforms. pLDDT – predicted Local Distance Difference test estimates per-residue modelling confidence on the scale from 0 to 100 (lowest to highest). F – submitted compute job failed to complete due to computational/time limits of the web server. 4. DISCUSSION Seven discussed rare mutations in this work have one supporting (ACMG PP3) and one moderate (ACMG PM2) evidence of pathogenicity. Only one out of seven has conflicting evidence of a benign effect (ACMG BS2). Each of these mutations was unique to a single family and was not observed in other families. Some were observed at higher frequencies in the South Asian population in general. Consequently, each identified mutation was placed in seven genes: SRD5A3 , RDH11 , RTF2 , PCDHA2 , ADAMTS17 , UNC13B , and TRPC3 . Only the SRD5A3 gene was previously linked to ID [ 47 – 49 ] and is included in the approved gene panel for ID (v5.0 from 22 March 2023) available at Genomics England’s PanelApp [ 5 ]. The rs754837473 A>G (p.Tyr169Cys) missense mutation inside SRD5A3 identified in this study was also reported as of uncertain significance by one submitter in ClinVar in an individual with overlapping phenotype. The in silico protein structure modelling suggested changes in protein structure caused by this mutation. This evidence strongly suggests that rs754837473 A>G (p.Tyr169Cys) is a likely cause of ID. Despite this, according to the ACMG guidance, more evidence from functional studies may be needed to re-classify it as pathogenic. Three identified genes, RDH11 , ADAMTS17 , and TRPC3 , were not previously implicated in ID but were linked to other rare conditions, which include RDJCSS (MIM: 616108), WMS4 (MIM: 613195), and SCA41 (MIM: 616410), respectively. For the two other genes, PCDHA2 and UNC13B, also not known to cause ID, previous studies reported their high expression in the central nervous system [ 66 ] and cerebral cortex [ 67 ], respectively. Lastly, two recent studies have reported the association between the increased levels of RTF2 and the risk of Alzheimer’s disease. Taking this on board, we can state that these six genes, not previously known to cause ID, merit further investigations. Genotyping additional family members, which is one of the subjects in our future work, may help us prioritize RDH11 vs RTF2 and PCDHA2 vs ADAMTS17 , which are pairs of the genes observed within the same family. However, without functional studies, additional evidence of co-segregation will not be sufficient to re-classify observed mutation from uncertain significance to pathogenic. The findings presented here contribute novel and highly needed information for interpreting rare mutations in genetic testing of ID and provide a starting point for further functional studies of novel candidate mutations and genes. 5. CONTRIBUTIONS A.W. collected samples, performed data analyses and wrote the manuscript. R.E., D.P., and J.S. generated and processed sequencing data. F.L. oversaw the sequencing of samples. Z.M. and A.M. provided help with sample collection and financial support. K.T., R.C.M., and C.B. provided critical feedback on the manuscript. G.A. and A.H.K. supervised the Ph.D. research of A.W. V.M. oversaw the sequencing study, and provided critical feedback on the manuscript. D.T. supervised A.W. in data analyses and wrote the manuscript. All authors read the manuscript. 7. COMPETING INTERESTS The authors have no relevant financial or non-financial interests to disclose. 8. DATA AVAILABILITY Data will be made available on reasonable request, which can be directed to Abdul Waheed and Daniel Taliun. SUPPLEMENTARY TABLES AND FIGURES View this table: View inline View popup Download powerpoint Supplementary Table 1. Depth of coverage and estimated contamination rates for individuals in AR ID families. FID – family ID. IID – individual ID. DP – depth of coverage. The sample contamination is a Freemix score computed using VerifyBamId and the HGDP reference panel. * - affected individual. View this table: View inline View popup Download powerpoint Supplementary Table 2. The number of called SNVs and indels per individual. FID – family ID. IID – individual ID. CNN1D and CNN2D are based on Convolutional Neural Network (CNN) scores implemented in GATK CNNScoreVariants. A consensus filter is an intersect of variants which pass all three filters: hard, CNN1D, and CNN2D. * - affected individual. View this table: View inline View popup Download powerpoint Supplementary Table 3. Total number of variants across all sequenced individuals stratified by variant type, deleteriousness, and frequency. When merging variants from multiple individuals, only those variant positions were kept, which had DP > 10 across all individuals. Population alternate allele frequencies are based on gnomAD and 1000 Genomes Project. The variant was predicted deleterious if it had PolyPhen > 0.15 or SIFT 20. View this table: View inline View popup Download powerpoint Supplementary Table 4. Multiple alignments of 10 primates at 4:55364215:A/G (rs754837473) inside the SRD5A3 gene. The ancestral allele is marked in bold and underscored. Gene orthologues – 1-to-1(one gene in humans is orthologous to one gene in another species) or 1-to-M (one gene in humans is orthologous to multiple genes in another species).WGA score – Ensembl’s Whole Genome Alignment score between two orthologues. GOC score – Ensembl’s Genome Order Conservation score between two orthologues. Identity – percent of human sequence matching sequence of the orthologue. If human gene has multiple orthologues, then the orthologue with the best WGS, GOC, and identity scores is reported. View this table: View inline View popup Download powerpoint Supplementary Table 5. Constraint metrics from the gnomAD v2.1.1, gnomAD v4.0.0, and DECIPHER databases. oe – observed vs. expected ratio. LoF – loss of function. CI – confidence interval. The oe LoF, oe missense, and missense Z-score metrics were reported in the gnomAD v2.1.1 database; the missense intolerance score (γ) metric was reported in the DECIPHER database. The missense intolerance score is reported for the region, which overlaps the corresponding mutation within a gene. Values in bold satisfy the recommended intolerance thresholds: missense Z-score >3.09; missense intolerance score (γ) P-value < 0.001. View this table: View inline View popup Download powerpoint Supplementary Table 6. Multiple alignments of 10 primates at 14:67692437:C/T inside the RDH11 gene. See Supplementary Table 4 for definitions. View this table: View inline View popup Download powerpoint Supplementary Table 7. Multiple alignments of 10 primates at 20:56473304: G/A (rs530872363) inside the RTF2 gene. See Supplementary Table 4 for definitions. View this table: View inline View popup Download powerpoint Supplementary Table 8. Multiple alignments of 10 primates at 5:140796317:C/A (rs376199248) inside the PCDHA2 gene. See Supplementary Table 4 for definitions. View this table: View inline View popup Download powerpoint Supplementary Table 9. Multiple alignments of 10 primates at 15:100331016:G/C inside the ADAMTS17 gene. See Supplementary Table 4 for definitions. View this table: View inline View popup Download powerpoint Supplementary Table 10. Multiple alignments of 10 primates at 9:35403770:C/T (rs142697917) inside the UNC13B gene. See Supplementary Table 4 for definitions. View this table: View inline View popup Download powerpoint Supplementary Table 11. Multiple alignments of 10 primates at 4:121902938: T/A (rs145430158) inside the TRPC3 gene. See Supplementary Table 4 for definitions. Download figure Open in new tab Supplementary Figure 1. The districts (coloured) of Azad Jammu and Kashmir, Pakistan, where participants were recruited. Download figure Open in new tab Supplementary Figure 2. Single sample variant calling and filtering pipeline using GATK. Download figure Open in new tab Supplementary Figure 3. Pipeline for merging single sample VCFs. Download figure Open in new tab Supplementary Figure 4. Pipeline for mapping autosomal recessive variants for ID. Download figure Open in new tab Supplementary Figure 5. Coverage and depth in sequenced individuals with (affected) and without (unaffected) ID. DP – depths of coverage. The box bounds the IQR, and Tukey-style whiskers extend to a maximum of 1.5 ⨉ IQR beyond the box. The horizontal line within the box indicates the median value. Open circles represent individual data points. Download figure Open in new tab Supplementary Figure 6. Normalized depth of coverage on sex chromosomes in sequenced individuals. Open circles represent sequenced individuals reported as females. Open squares represent sequenced individuals reported as males. Download figure Open in new tab Supplementary Figure 7. PCA projection of the sequenced individuals into Human Genome Diversity Project (HGDP) reference panel. Red star symbols represent sequenced individuals. Colored circles represent reference individuals from the HGDP dataset. Download figure Open in new tab Supplementary Figure 8. The number of variants in affected and unaffected sequenced individuals. Variants include SNVs and indels, which passed all quality checks. Left panel: the box bounds the IQR, and Tukey-style whiskers extend to a maximum of 1.5 ⨉ IQR beyond the box; the horizontal line within the box indicates the median value; open circles represent individual data points; the p-value corresponds to the two-tailed Mann-Whitney U test. Right panel: DP – depths of coverage; open diamonds represent the number of variants and average sequencing depth in unaffected individuals; shaded triangles represent the number of variants and average sequencing depth in individuals affected with intellectual disability. 6. ACKNOWLEDGMENTS A.W. was supported by the Higher Education Commission of Pakistan. The sequencing experiments and analyses of sequencing data in this study were supported by the McGill Canada Excellence Research Chair Program in Genomic Medicine (V.M.). V.M. holds a Canada Excellence Research Chair. Calcul Quebec and the Digital Research Alliance of Canada provided the computational resources for this study. The authors would like to thank Claire Le Moigne from the McGill Canada Excellence Research Chair Program in Genomic Medicine for administrative support. REFERENCES 1. ↵ McKenzie , K. , et al. , Systematic review of the prevalence and incidence of intellectual disabilities: current trends and issues . Current Developmental Disorders Reports , 2016 . 3 : p. 104 – 115 . OpenUrl 2. Patel , D.R. , et al. , A clinical primer on intellectual disability . Translational Pediatrics , 2020 . 9 ( S1 ): p. S23 – S35 . OpenUrl 3. ↵ American Psychiatric , A., Diagnostic and Statistical Manual of Mental Disorders . 2022 . 4. ↵ Vissers , L.E.L.M. , C. Gilissen , and J.A. Veltman , Genetic studies in intellectual disability and related disorders . Nature Reviews Genetics , 2015 . 17 ( 1 ): p. 9 – 18 . OpenUrl CrossRef 5. ↵ Martin , A.R. , et al. , PanelApp crowdsources expert knowledge to establish consensus diagnostic gene panels . Nature Genetics , 2019 . 51 ( 11 ): p. 1560 – 1565 . OpenUrl CrossRef PubMed 6. ↵ Ilyas , M. , et al. , The genetics of intellectual disability: advancing technology and gene editing . F1000Research , 2020 . 9 . 7. ↵ Jansen , S. , L.E.L.M. Vissers , and B.B.A. de Vries , The Genetics of Intellectual Disability . Brain Sciences , 2023 . 13 ( 2 ). 8. ↵ Bennett , R.L. , et al. , Standardized Human Pedigree Nomenclature: Update and Assessment of the Recommendations of the National Society of Genetic Counselors . Journal of Genetic Counseling , 2008 . 17 ( 5 ): p. 424 – 433 . OpenUrl CrossRef PubMed 9. ↵ Bourgey , M. , et al. , GenPipes: an open-source framework for distributed and scalable genomic analyses . GigaScience , 2019 . 8 ( 6 ). 10. ↵ Zhang , F. , et al. , Ancestry-agnostic estimation of DNA sample contamination from sequence reads . Genome Research , 2020 . 30 ( 2 ): p. 185 – 194 . OpenUrl Abstract / FREE Full Text 11. ↵ Van der Auwera GA and O’Connor BD , Genomics in the Cloud: Using Docker, GATK, and WDL in Terra (1st Edition) . 2020 : O’Reilly Media . 12. ↵ Danecek , P. , et al. , Twelve years of SAMtools and BCFtools . GigaScience , 2021 . 10 ( 2 ). 13. ↵ McLaren , W. , et al. , The Ensembl Variant Effect Predictor . Genome Biology , 2016 . 17 ( 1 ). 14. ↵ Chen , S. , et al. , A genome-wide mutational constraint map quantified from variation in 76,156 human genomes . bioRxiv , 2022 . 15. ↵ Auton , A. , et al. , A global reference for human genetic variation . Nature , 2015 . 526 ( 7571 ): p. 68 – 74 . OpenUrl CrossRef PubMed 16. ↵ Landrum , M.J. , et al. , ClinVar: improving access to variant interpretations and supporting evidence . Nucleic Acids Research , 2018 . 46 ( D1 ): p. D1062 – D1067 . OpenUrl CrossRef PubMed 17. ↵ Felsenstein , J. and G.A. Churchill , A Hidden Markov Model approach to variation among sites in rate of evolution . Molecular Biology and Evolution , 1996 . 13 ( 1 ): p. 93 – 104 . OpenUrl CrossRef PubMed Web of Science 18. Pollard , K.S. , et al. , Detection of nonneutral substitution rates on mammalian phylogenies . Genome Research , 2010 . 20 ( 1 ): p. 110 – 121 . OpenUrl Abstract / FREE Full Text 19. Siepel , A. , et al. , Evolutionarily conserved elements in vertebrate, insect, worm, and yeast genomes . Genome Research , 2005 . 15 ( 8 ): p. 1034 – 1050 . OpenUrl Abstract / FREE Full Text 20. Siepel , A. and D. Haussler , Phylogenetic Hidden Markov Models, in Statistical Methods in Molecular Evolution . 2005 . p. 325 – 351 . 21. ↵ Yang , Z ., A space-time process model for the evolution of DNA sequences . Genetics , 1995 . 139 ( 2 ): p. 993 – 1005 . OpenUrl Abstract / FREE Full Text 22. ↵ Kent , W.J. , et al. , The Human Genome Browser at UCSC . Genome Research , 2002 . 12 ( 6 ): p. 996 – 1006 . OpenUrl Abstract / FREE Full Text 23. ↵ Wasserman , W.W. , et al. , Identifying a High Fraction of the Human Genome to be under Selective Constraint Using GERP++ . PLoS Computational Biology , 2010 . 6 ( 12 ). 24. ↵ Cunningham , F. , et al. , Ensembl 2022 . Nucleic Acids Research , 2022 . 50 ( D1 ): p. D988 – D995 . OpenUrl CrossRef 25. ↵ Pejaver , V. , et al. , Calibration of computational tools for missense variant pathogenicity classification and ClinGen recommendations for PP3/BP4 criteria . The American Journal of Human Genetics , 2022 . 109 ( 12 ): p. 2163 – 2177 . OpenUrl CrossRef 26. ↵ Karczewski , K.J. , et al. , The mutational constraint spectrum quantified from variation in 141,456 humans . Nature , 2020 . 581 ( 7809 ): p. 434 – 443 . OpenUrl CrossRef PubMed 27. ↵ Samocha , K.E. , et al. , A framework for the interpretation of de novo mutation in human disease . Nature Genetics , 2014 . 46 ( 9 ): p. 944 – 950 . OpenUrl CrossRef PubMed 28. ↵ Samocha , K.E ., et al. , Regional missense constraint improves variant deleteriousness prediction . bioRxiv , 2017 . 29. ↵ Foreman , J. , et al. , DECIPHER: Supporting the interpretation and sharing of rare disease phenotype-linked variant data to advance diagnosis and research . Human Mutation , 2022 . 30. ↵ Delgado , J. , et al. , FoldX 5.0: working with RNA, small molecules and a new graphical interface . Bioinformatics , 2019 . 35 ( 20 ): p. 4168 – 4169 . OpenUrl 31. ↵ Uversky , V.N. , et al. , RosettaScripts: A Scripting Language Interface to the Rosetta Macromolecular Modeling Suite . PLoS ONE , 2011 . 6 ( 6 ). 32. ↵ Leaver-Fay , A. , et al. , Rosetta3 , in Computer Methods , Part C . 2011 . p. 545 – 574 . 33. ↵ Ferla , M.P. , et al. , Venus: Elucidating the Impact of Amino Acid Variants on Protein Function Beyond Structure Destabilisation . Journal of Molecular Biology , 2022 . 434 ( 11 ). 34. ↵ Ittisoponpisan , S. , et al. , Can Predicted Protein 3D Structures Provide Reliable Insights into whether Missense Variants Are Disease Associated? Journal of Molecular Biology , 2019 . 431 ( 11 ): p. 2197 – 2212 . OpenUrl CrossRef PubMed 35. ↵ Armstrong , D.R. , et al. , PDBe: improved findability of macromolecular structure data in the PDB . Nucleic Acids Research , 2019 . 36. ↵ Varadi , M. , et al. , AlphaFold Protein Structure Database: massively expanding the structural coverage of protein-sequence space with high-accuracy models . Nucleic Acids Research , 2022 . 50 ( D1 ): p. D439 – D444 . OpenUrl CrossRef PubMed 37. ↵ Park , H. , et al. , Simultaneous Optimization of Biomolecular Energy Functions on Features from Small Molecules and Macromolecules . Journal of Chemical Theory and Computation , 2016 . 12 ( 12 ): p. 6201 – 6212 . OpenUrl 38. ↵ Frenz , B. , et al. , Prediction of Protein Mutational Free Energy: Benchmark and Sampling Improvements Increase Classification Accuracy . Frontiers in Bioengineering and Biotechnology , 2020 . 8 . 39. ↵ Tyka , M.D. , et al. , Alternate States of Proteins Revealed by Detailed Energy Landscape Mapping . Journal of Molecular Biology , 2011 . 405 ( 2 ): p. 607 – 618 . OpenUrl CrossRef PubMed 40. Khatib , F. , et al. , Algorithm discovery by protein folding game players . Proceedings of the National Academy of Sciences , 2011 . 108 ( 47 ): p. 18949 – 18953 . OpenUrl Abstract / FREE Full Text 41. ↵ Maguire , J.B. , et al. , Perturbing the energy landscape for improved packing during computational protein design . Proteins: Structure, Function, and Bioinformatics , 2020 . 89 ( 4 ): p. 436 – 449 . OpenUrl 42. ↵ Ferla , M.P. , et al. , MichelaNglo: sculpting protein views on web pages without coding . Bioinformatics , 2020 . 36 ( 10 ): p. 3268 – 3270 . OpenUrl 43. ↵ Chaudhury , S. , S. Lyskov , and J.J. Gray , PyRosetta: a script-based interface for implementing molecular modeling algorithms using Rosetta . Bioinformatics , 2010 . 26 ( 5 ): p. 689 – 691 . OpenUrl CrossRef PubMed Web of Science 44. ↵ Alford , R.F. , et al. , The Rosetta All-Atom Energy Function for Macromolecular Modeling and Design . Journal of Chemical Theory and Computation , 2017 . 13 ( 6 ): p. 3031 – 3048 . OpenUrl 45. ↵ National Center for Biotechnology Information ClinVar [VCV002506565.2] . [cited 2023 November 29]; Available from: https://www.ncbi.nlm.nih.gov/clinvar/variation/VCV002506565.2 . 46. ↵ Backman , J.D. , et al. , Exome sequencing and analysis of 454,787 UK Biobank participants . Nature , 2021 . 599 ( 7886 ): p. 628 – 634 . OpenUrl CrossRef PubMed 47. ↵ Cantagrel , V. , et al. , SRD5A3 is required for converting polyprenol to dolichol and is mutated in a congenital glycosylation disorder . Cell , 2010 . 142 ( 2 ): p. 203 – 17 . OpenUrl CrossRef PubMed Web of Science 48. Kasapkara , C.S. , et al. , SRD5A3-CDG: a patient with a novel mutation . Eur J Paediatr Neurol , 2012 . 16 ( 5 ): p. 554 – 6 . OpenUrl CrossRef PubMed 49. ↵ Najmabadi , H. , et al. , Deep sequencing reveals 50 novel genes for recessive cognitive disorders . Nature , 2011 . 478 ( 7367 ): p. 57 – 63 . OpenUrl CrossRef PubMed Web of Science 50. ↵ Kahrizi , K. , et al. , Next generation sequencing in a family with autosomal recessive Kahrizi syndrome (OMIM 612713) reveals a homozygous frameshift mutation in SRD5A3 . Eur J Hum Genet , 2011 . 19 ( 1 ): p. 115 – 7 . OpenUrl CrossRef PubMed 51. ↵ Richards , S. , et al. , Standards and guidelines for the interpretation of sequence variants: a joint consensus recommendation of the American College of Medical Genetics and Genomics and the Association for Molecular Pathology . Genetics in Medicine , 2015 . 17 ( 5 ): p. 405 – 424 . OpenUrl CrossRef PubMed 52. ↵ Jaganathan , K. , et al. , Predicting Splicing from Primary Sequence with Deep Learning . Cell , 2019 . 176 ( 3 ): p. 535 – 548 .e24. OpenUrl CrossRef PubMed 53. ↵ Xie , Y. , et al. , New syndrome with retinitis pigmentosa is caused by nonsense mutations in retinol dehydrogenase RDH11 . Human Molecular Genetics , 2014 . 23 ( 21 ): p. 5774 – 5780 . OpenUrl CrossRef PubMed 54. ↵ Wingo , A.P. , et al. , Integrating human brain proteomes with genome-wide association data implicates new proteins in Alzheimer’s disease pathogenesis . Nature Genetics , 2021 . 53 ( 2 ): p. 143 – 146 . OpenUrl CrossRef 55. ↵ Ou , Y.-N. , et al. , Identification of novel drug targets for Alzheimer’s disease by integrating genetics and proteomes from brain and blood . Molecular Psychiatry , 2021 . 26 ( 10 ): p. 6065 – 6073 . OpenUrl 56. ↵ Weissbach , A. , et al. , Exome sequencing in a family with restless legs syndrome . Movement disorders , 2012 . 27 ( 13 ): p. 1686 – 1689 . OpenUrl PubMed 57. ↵ Kaminsky , E.B. , et al. , An evidence-based approach to establish the functional and clinical significance of copy number variants in intellectual and developmental disabilities . Genetics in medicine , 2011 . 13 ( 9 ): p. 777 – 784 . OpenUrl CrossRef PubMed 58. ↵ Miller , D.T. , et al. , Consensus statement: chromosomal microarray is a first-tier clinical diagnostic test for individuals with developmental disabilities or congenital anomalies . Am J Hum Genet , 2010 . 86 ( 5 ): p. 749 – 64 . OpenUrl CrossRef PubMed Web of Science 59. ↵ Shah , M.H. , et al. , Whole exome sequencing identifies a novel splice-site mutation in ADAMTS17 in an Indian family with Weill-Marchesani syndrome . Molecular vision , 2014 . 20 : p. 790 . OpenUrl PubMed 60. ↵ Morales , J. , et al. , Homozygous mutations in ADAMTS10 and ADAMTS17 cause lenticular myopia, ectopia lentis, glaucoma, spherophakia, and short stature . The American Journal of Human Genetics , 2009 . 85 ( 5 ): p. 558 – 568 . OpenUrl CrossRef PubMed Web of Science 61. ↵ Marzin , P. , V. Cormier-Daire , and E. Tsilou, Weill-marchesani syndrome . 2020 . 62. ↵ Wang , J. , et al. , UNC13B variants associated with partial epilepsy with favourable outcome . Brain , 2021 . 144 ( 10 ): p. 3050 – 3060 . OpenUrl 63. ↵ Becker , E.B. , et al. , A point mutation in TRPC3 causes abnormal Purkinje cell development and cerebellar ataxia in moonwalker mice . Proceedings of the National Academy of Sciences , 2009 . 106 ( 16 ): p. 6706 – 6711 . OpenUrl Abstract / FREE Full Text 64. ↵ Becker , E.B.E. , et al. , Candidate Screening of the TRPC3 Gene in Cerebellar Ataxia . The Cerebellum , 2011 . 10 ( 2 ): p. 296 – 299 . OpenUrl 65. ↵ Fogel , B.L. , S.M. Hanson , and E.B. Becker , Do mutations in the murine ataxia gene TRPC3 cause cerebellar ataxia in humans? Movement disorders: official journal of the Movement Disorder Society , 2015 . 30 ( 2 ): p. 284 . OpenUrl 66. ↵ Mancini , M. , S. Bassani , and M. Passafaro , Right Place at the Right Time: How Changes in Protocadherins Affect Synaptic Connections Contributing to the Etiology of Neurodevelopmental Disorders . Cells , 2020 . 9 ( 12 ). 67. ↵ Wang , J. , et al. , UNC13B variants associated with partial epilepsy with favourable outcome . Brain , 2021 . 144 ( 10 ): p. 3050 – 3060 . OpenUrl View the discussion thread. Back to top Previous Next Posted January 10, 2024. Download PDF Email Thank you for your interest in spreading the word about medRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following Rare splice and missense variants with evidence of pathogenicity in consanguineous families with autosomal recessive intellectual disability from Pakistan Message Subject (Your Name) has forwarded a page to you from medRxiv Message Body (Your Name) thought you would like to see this page from the medRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share Rare splice and missense variants with evidence of pathogenicity in consanguineous families with autosomal recessive intellectual disability from Pakistan Abdul Waheed , Robert Eveleigh , Danielle Perley , Janick St-Cyr , François Lefebvre , Abdul Hameed Khan , Zarqash Majeed , Abrish Majeed , Katerina Trajanoska , Raquel Cuella-Martin , Claude Bhérer , Ghazanfar Ali , Vincent Mooser , Daniel Taliun medRxiv 2024.01.08.23299914; doi: https://doi.org/10.1101/2024.01.08.23299914 Share This Article: Copy Citation Tools Rare splice and missense variants with evidence of pathogenicity in consanguineous families with autosomal recessive intellectual disability from Pakistan Abdul Waheed , Robert Eveleigh , Danielle Perley , Janick St-Cyr , François Lefebvre , Abdul Hameed Khan , Zarqash Majeed , Abrish Majeed , Katerina Trajanoska , Raquel Cuella-Martin , Claude Bhérer , Ghazanfar Ali , Vincent Mooser , Daniel Taliun medRxiv 2024.01.08.23299914; doi: https://doi.org/10.1101/2024.01.08.23299914 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Genetic and Genomic Medicine Subject Areas All Articles Addiction Medicine (573) Allergy and Immunology (865) Anesthesia (304) Cardiovascular Medicine (4457) Dentistry and Oral Medicine (445) Dermatology (383) Emergency Medicine (610) Endocrinology (including Diabetes Mellitus and Metabolic Disease) (1517) Epidemiology (15244) Forensic Medicine (30) Gastroenterology (1132) Genetic and Genomic Medicine (6620) Geriatric Medicine (669) Health Economics (1002) Health Informatics (4557) Health Policy (1372) Health Systems and Quality Improvement (1615) Hematology (543) HIV/AIDS (1272) Infectious Diseases (except HIV/AIDS) (15936) Intensive Care and Critical Care Medicine (1106) Medical Education (624) Medical Ethics (147) Nephrology (670) Neurology (6635) Nursing (346) Nutrition (999) Obstetrics and Gynecology (1148) Occupational and Environmental Health (957) Oncology (3348) Ophthalmology (980) Orthopedics (369) Otolaryngology (421) Pain Medicine (436) Palliative Medicine (130) Pathology (665) Pediatrics (1696) Pharmacology and Therapeutics (693) Primary Care Research (714) Psychiatry and Clinical Psychology (5463) Public and Global Health (9257) Radiology and Imaging (2210) Rehabilitation Medicine and Physical Therapy (1371) Respiratory Medicine (1198) Rheumatology (598) Sexual and Reproductive Health (716) Sports Medicine (532) Surgery (714) Toxicology (99) Transplantation (289) Urology (265) (function(){function c(){var b=a.contentDocument||a.contentWindow.document;if(b){var d=b.createElement('script');d.innerHTML="window.__CF$cv$params={r:'a034beeadfa74193',t:'MTc4MDA0OTgwOQ=='};var a=document.createElement('script');a.src='/cdn-cgi/challenge-platform/scripts/jsd/main.js';document.getElementsByTagName('head')[0].appendChild(a);";b.getElementsByTagName('head')[0].appendChild(d)}}if(document.body){var a=document.createElement('iframe');a.height=1;a.width=1;a.style.position='absolute';a.style.top=0;a.style.left=0;a.style.border='none';a.style.visibility='hidden';document.body.appendChild(a);if('loading'!==document.readyState)c();else if(window.addEventListener)document.addEventListener('DOMContentLoaded',c);else{var e=document.onreadystatechange||function(){};document.onreadystatechange=function(b){e(b);'loading'!==document.readyState&&(document.onreadystatechange=e,c())}}}})();
Text is read by the "Ask this paper" AI Q&A widget below.
Extraction quality varies by source — PMC NXML preserves structure
cleanly, OA-HTML may include some navigation residue, and OA-PDF can
have broken hyphenation. The publisher copy
(via DOI)
is the canonical version.