Encoding of pretrained large language models mirrors the genetic architectures of human psychological traits

preprint OA: closed CC-BY-4.0
📄 Open PDF Full text JSON View at publisher

Abstract

Recent advances in large language models (LLMs) have prompted a frenzy in utilizing them as universal translators for biomedical terms. However, the black box nature of LLMs has forced researchers to rely on artificially designed benchmarks without understanding what exactly LLMs encode. We demonstrate that pretrained LLMs can already explain up to 51% of the genetic correlation between items from a psychometrically-validated neuroticism questionnaire, without any fine-tuning. For psychiatric diagnoses, we found disorder names aligned better with genetic relationships than diagnostic descriptions. Our results indicate the pretrained LLMs have encodings mirroring genetic architectures. These findings highlight LLMs’ potential for validating phenotypes, refining taxonomies, and integrating textual and genetic data in mental health research.
Full text 31,017 characters · extracted from preprint-html · click to expand
Encoding of pretrained large language models mirrors the genetic architectures of human psychological traits | medRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-P4HH5NV'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search Encoding of pretrained large language models mirrors the genetic architectures of human psychological traits Bohan Xu , Nick Obradovich , Wenjie Zheng , Robert Loughnan , Lucy Shao , Masaya Misaki , Wesley K. Thompson , Martin Paulus , View ORCID Profile Chun Chieh Fan doi: https://doi.org/10.1101/2025.03.27.25324744 Bohan Xu 1 Laureate Institute for Brain Research , Tulsa, Oklahoma, USA 2 Center for Population Neuroscience and Genetics, Laureate Institute for Brain Research , Tulsa, Oklahoma, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site Nick Obradovich 1 Laureate Institute for Brain Research , Tulsa, Oklahoma, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site Wenjie Zheng 3 Department of Radiology, University of California , La Jolla, California, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site Robert Loughnan 2 Center for Population Neuroscience and Genetics, Laureate Institute for Brain Research , Tulsa, Oklahoma, USA 4 Center for Human Development, University of California , La Jolla, California, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site Lucy Shao 5 Biostatistic Program, University of California , La Jolla, California, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site Masaya Misaki 1 Laureate Institute for Brain Research , Tulsa, Oklahoma, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site Wesley K. Thompson 1 Laureate Institute for Brain Research , Tulsa, Oklahoma, USA 2 Center for Population Neuroscience and Genetics, Laureate Institute for Brain Research , Tulsa, Oklahoma, USA 6 Herbert Wertheim School of Public Health & Human Longevity Science, University of California , La Jolla, California, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site Martin Paulus 1 Laureate Institute for Brain Research , Tulsa, Oklahoma, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site Chun Chieh Fan 1 Laureate Institute for Brain Research , Tulsa, Oklahoma, USA 2 Center for Population Neuroscience and Genetics, Laureate Institute for Brain Research , Tulsa, Oklahoma, USA 3 Department of Radiology, University of California , La Jolla, California, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Chun Chieh Fan For correspondence: cfan{at}laureateinstitute.org Abstract Full Text Info/History Metrics Data/Code Preview PDF Abstract Recent advances in large language models (LLMs) have prompted a frenzy in utilizing them as universal translators for biomedical terms. However, the black box nature of LLMs has forced researchers to rely on artificially designed benchmarks without understanding what exactly LLMs encode. We demonstrate that pretrained LLMs can already explain up to 51% of the genetic correlation between items from a psychometrically-validated neuroticism questionnaire, without any fine-tuning. For psychiatric diagnoses, we found disorder names aligned better with genetic relationships than diagnostic descriptions. Our results indicate the pretrained LLMs have encodings mirroring genetic architectures. These findings highlight LLMs’ potential for validating phenotypes, refining taxonomies, and integrating textual and genetic data in mental health research. Introduction Large language models (LLMs) have demonstrated remarkable advancements in recent years 1 , fueling enthusiasm for their potential as versatile tools across a variety of domains, including biomedical and clinical research. These models, trained on vast corpora of text, have been proposed for tasks ranging from extracting and linking information from electronic medical records to generating embeddings that predict clinical outcomes 2 , 3 . Yet, fundamental questions remain: What do LLMs truly encode within their latent spaces? And how can we rigorously benchmark these representations at scale? To explore these questions, we examined whether LLMs can implicitly capture biologically grounded relationships among psychological and psychiatric constructs. Specifically, we focused on the “zero-shot” performance of LLM embeddings—how well they reflect relationships between concepts without fine-tuning. To provide a biologically informed benchmark, we leveraged population-level genetic correlations (GCs) 4 , which quantify the shared effects of single nucleotide polymorphisms (SNPs) between traits or disorders. GCs offer a robust measure of biological relatedness, making them an ideal reference for evaluating whether LLM embeddings capture meaningful patterns 5 , 6 . We specifically probed: To which degree, without additional training, the LLM-based textual embeddings of items from a neuroticism questionnaire and DSM-5 diagnostic categories reflect their genetically defined relationships? The convergence of textual embeddings with GCs would indicate that LLMs are not merely encoding arbitrary or idiosyncratic semantic associations but instead are aligning with human biology. In contrast to the frequently used text retrieval benchmarks 7 , which evaluate the accuracies of the text queries given manually designed artificial datasets, our results demonstrate the potential to bridge two disparate knowledge domains—textual corpora and genetic data—shedding light on how statistical patterns in language reflect, and sometimes mirror, the biological organization of traits and disorders. Results Examining the alignment between text embedding and biological embedding for Neuroticism questionnaire items We selected 10 different LLM text embedding models, eight of them being state-of-the-art encoder models (as of this writing) available in recent AI booms, with the other two being popular open-source models (detailed in Supplementary Table 1 ). Detailed procedures of the text embedding, as well as the comparisons with the estimated GCs can be found in the Online Methods . Figure 1a illustrates the procedural steps we used to compare the text embeddings from 10 different LLM models with the biological embeddings based on genetic correlations estimated using a population cohort. The resulting performance metric, i.e. how much of the variation in GCs between 12 items of a neuroticism questionnaire ( Supplementary Table 2 ) 8 can be explained by zero-shot text embeddings of those questionnaire items, is shown in the Figure 1b and Figure 1c . View this table: View inline View popup Download powerpoint Supplementary Table 1. View this table: View inline View popup Supplementary Table 2. Download figure Open in new tab Figure 1. Comparisons between text embedding and genetically defined biological embedding. a). Conceptual framework of current study. b). Example of the alignment between text embedding and genetic correlations. The text embedding was derived using OpenAI 3-large model. Each point represents one pair of the questionnaire items. 95% confidence intervals of the genetic correlation estimates were provided. c). Variance of genetic correlations explained by zero-shot text embedding from the LLMs, estimated with linear mixed effects meta-regression. Colored by the model groups. Figure 1b showcases a detailed view of the text embeddings derived from the OpenAI’s v3 large-dimension model, demonstrating remarkable alignment with the empirical GCs of neuroticism items estimated from large-scale population cohort 8 , 9 . Notably, the embeddings captured the clustering of specific items that share stronger heritable relationships, such as “mood swings” and “nervousness,” which exhibited high genetic correlation in the UK Biobank data. The robustness of this alignment is underscored by the absence of major outliers, even when meta-regressions accounting for uncertainty in the genetic estimates ( Figure 1b ). The neuroticism questionnaire’s well-validated psychometric properties 8 and high-powered genetic estimates (N ≈ 84,000–100,000 for item-level GWAS) likely contributed to the strong alignment observed. This suggests that the semantic relationships encoded by LLMs reflect the latent structure of how individuals respond to these items in a biologically meaningful way. Across 10 selected LLMs, larger models, particularly those without dimensionality reduction, demonstrated a clear performance advantage over smaller models ( Figure 1c ). For instance, OpenAI’s v3 large-dimension model captured 51% of the variance of the genetic embeddings, compared to 37% for the dimensionally reduced v3 model and 27% for the older ada v2 model ( Figure 1c ). This progressive improvement highlights the importance of model scale and dimensionality in encoding biologically grounded relationships. Psychiatric Disorders: Disorder Names vs. DSM-5 Criteria When examining the six major psychiatric disorders (Major Depressive Disorder, Generalized Anxiety Disorder, Bipolar disorder, Schizophrenia, Attention Deficit and Hyperactivity Disorder, and Post-Traumatic Stress Disorder), we found that the text embeddings of disorder names captured genetic relationships more effectively than embeddings based on the descriptive criteria A of DSM-V ( Figure 2a and 2b ). Embedding of the DSM-5 descriptions showed miniscule alignment ( Figure 2a ), whereas the text embedding based on the diagnostic name only capture up to 29% of the variance of the GCs ( Figure 2b ). The superior performance of disorder names suggests that the textual data from which LLMs derive their embeddings may better capture co-occurrence patterns and conceptual associations in real-world contexts than does the semantic information from the diagnostic descriptions. This observation is consistent with the idea that real-world discourse more accurately reflects underlying biological relationships, perhaps because it implicitly incorporates relationships between comorbidities and shared etiological pathways. Download figure Open in new tab Figure 2. Variance of genetic correlations between psychiatric disorders explained by zero-shot text embedding from LLMs. a). Results based on encoding the descriptions from DSM-V diagnostic criteria. b). Results based on encoding the diagnostic names from DSM-V, e.g. schizophrenia and major depressive disorder. Discussion Our findings demonstrate that pretrained encoder-only large language models (LLMs) can approximate biologically grounded relationships among psychological traits and psychiatric disorders with surprising fidelity, even in a zero-shot setting. These models, trained solely on vast text corpora, encode emergent properties in their semantic spaces that reflect population-level patterns observed in genetic correlations (GCs). For example, embeddings from the largest OpenAI model tested captured 51% of the variance in the GC structure for neuroticism items, suggesting that LLMs are capable of encoding meaningful relationships rooted in human biology. This study highlights the potential of population genetic data to serve as a novel benchmark for evaluating LLM embeddings. Unlike traditional benchmarks based on human annotations or curated tasks, genetic correlations provide large-scale, biologically grounded indices that are relatively unbiased by selection processes. The alignment between LLM embeddings and GCs suggests that these models can uncover latent structures in language that mirror real-world, biologically meaningful relationships. Such benchmarks could enable systematic evaluations of how well LLMs represent relationships across domains, moving beyond tasks focused solely on linguistic accuracy or surface-level semantics. A key observation is that LLMs more effectively capture genetic relationships when encoding disorder names than when using DSM-5 diagnostic criteria. This finding suggests that the statistical patterns in real-world language use—across scientific articles, medical forums, and general texts— reflect an implicit taxonomy of psychological and psychiatric constructs. This taxonomy, shaped by how humans naturally discuss and group concepts, may already encode partial knowledge of their genetic and neurobiological relationships. In contrast, the weaker alignment of DSM-5 criteria highlights potential limitations in formal diagnostic systems. These criteria are linguistically complex, often overlapping, and rooted in consensus rather than strictly biological underpinnings. The discrepancy underscores the importance of reconsidering how psychiatric knowledge is formalized and communicated, particularly in ways that might better align with emerging biological insights. These findings mark a significant step toward integrating textual data with biological insights to advance mental health research. The convergence of LLM embeddings with population-level genetic correlations reveals the potential of AI systems to uncover meaningful relationships in human health and behavior. This cross-disciplinary approach holds promise for refining psychiatric taxonomies, developing personalized medicine tools, and inspiring new methodologies for questionnaire design and phenotypic discovery. Moving forward, the ability of LLMs to highlight undiscovered or underexplored biological relationships offers a unique opportunity for hypothesis generation and refinement of diagnostic frameworks. By systematically identifying trait or disorder pairs with high semantic similarity but little-known genetic overlap, researchers could conduct targeted genome-wide association studies to confirm potential shared heritability. Such zero-shot insights may also spur the development of more biologically informed psychiatric taxonomies that align with how disorders are discussed in natural language, moving beyond formal diagnostic criteria that often struggle to capture genetic and neurobiological subtleties. Iterative validation in prospective clinical studies—integrating patient risk profiles, real-world discourse, and genetic data—could then help validate these newly discovered relationships, ensuring that any resultant shifts in nosology or treatment guidelines are both biologically grounded and clinically applicable. Data Availability All data produced in the present work are contained in the manuscript Online Methods Items used in the embedding queries We selected a Neuroticism questionnaire as our main text embedding query due to their psychometric properties and readily available large-scale genome-wide association results 8 , 10 . Genetic correlations are likely more accurate given a well powered study design. The neuroticism questionnaire was made up of 12 items from the Eysenck Personality Questionnaire, Revised Short Form (EPQ-RS) 11 - the exact text of the questionnaire items used in this study can be found in Supplementary Table 1 . Six psychiatric disorders which have genetic correlations estimated in the same genome-wide association study 8 , 10 are included in our analyses, which are Major depressive disorder, Schizophrenia, Anorexia Nervosa, Autism, Bipolar I Disorder, and Attention-Deficit/Hyperactivity Disorder. We derived their text embedding based on either their descriptive criteria from Diagnostic and Statistical Manual of Mental Disorders, Fifth Edition, Text Revision or their diagnostic name only. Detailed text fed into the LLMs can also be found in Supplementary Table 1 . Encoder-only Large Language Models Unlike encoder-decoder and decoder-only models designed for generative tasks like ChatGPT, embedding models (encoders) focus on transforming input sentence, into a meaningful numerical representation known as an embedding. While all these models utilize the attention mechanism 12 , their primary differences are the architecture and pretraining tasks 13 – 16 . Decoders employ masked attention, restricting access to only earlier positions in the output sequence, which enables sequential generation. In contrast, encoders process the entire input sequence simultaneously, capturing holistic contextual information. Decoders are typically pretrained using Causal Language Modeling (CLM), which involves predicting the next token (the fundamental unit in LLMs) in a sequence based on prior tokens 14 , 15 . In contrast, encoders commonly use Masked Language Modeling (MLM) for pretraining, where tokens in the input sequence are randomly masked, and the model is trained to predict these masked tokens 13 , 16 . In the following sections, we described the encoder-only models we applied in this study. Open-source models Bidirectional Encoder Representations from Transformers (BERT) 13 is the first model to leverage the self-attention mechanism while jointly conditioning on both left and right context, for creating vector representations of language. The initial release of BERT includes two variants, where BERT BASE has about 110M parameters while BERT LARGE has approximately 340M parameters. Both models are pre-trained with same data, BooksCorpus and English Wikipedia, as well as same tasks, MLM and Next Sentence Prediction (NSP). Liu et al. 16 introduced RoBERTa (Robustly Optimized BERT Approach), a pretraining and optimization strategy designed to enhance the performance of the BERT architecture. RoBERTa replaces the WordPiece tokenizer used in the original BERT with Byte-Pair Encoding (BPE) for generating the token vocabulary. It eliminates the NSP task from pretraining and modifies the MLM task by incorporating dynamic masking. Additionally, RoBERTa leverages a significantly larger dataset, longer training duration, larger batch sizes, and an optimized learning rate schedule to achieve superior performance. Closed-source models Compared to open-source models, closed-source models generally provide more powerful solutions for generating embeddings. This is because they are often trained with large-scale, high-quality proprietary datasets and leverage advanced architectures and training techniques that are not publicly accessible. The superiority of those models compared to older generation open sourced models, such as BERT and RoBERTa, is evident in the text retrieval benchmark leaderboard, MTEB 7 , 17 . In this study, we chose the encoder-only models from Cohere, Google, and OpenAI ( Supplementary Table 2 ). However, despite those models being publicly available for deployment through web interfaces, the proprietary nature of those models makes some parameters less transparent, hence difficult to compare. Therefore, we choose models from those three companies based on the versioning provided ( Supplementary Table 2 ). Comparing Encoded Spaces The distances between encoded texts are defined by the cosine similarities of the embedded values. For each pair of the texts, the correlations between two vectors of the embedded values are calculated. The pairwise distances of the input texts inform us about the relative positions of each item in the embedded space. By comparing the distances in the text embeddings (cosine similarities) and the distances in the biological embeddings (genetic correlations), we can evaluate how well those two embeddings align with each other. We used linear mixed effects meta-analysis to evaluate the degree of the alignment between two types of the embeddings. Acknowledgement This work was partly funded by The William K. Warren Foundation, the National Institute of General Medical Sciences Center (Grant 2 P20GM121312, MPP, RK, KLF), the National Institute on Drug Abuse (U01DA050989, MPP), and the National Institute of Mental Health (R01MH122688, R01MH128959, CCF). Dr. Paulus advises Spring Care, Inc., receives royalties from an article on methamphetamine in UpToDate, and has a compensated consulting agreement with Boehringer Ingelheim International GmbH. References ↵ Moor , M. et al. Foundation models for generalist medical artificial intelligence . Nature 616 , 259 – 265 ( 2023 ). doi: 10.1038/s41586-023-05881-4 OpenUrl CrossRef PubMed ↵ Jiang , L. Y. et al. Health system-scale language models are all-purpose prediction engines . Nature ( 2023 ). doi: 10.1038/s41586-023-06160-y OpenUrl CrossRef ↵ Luo , R. et al. BioGPT: generative pre-trained transformer for biomedical text generation and mining . Briefings in Bioinformatics 23 ( 2022 ). doi: 10.1093/bib/bbac409 OpenUrl CrossRef ↵ van Rheenen , W. , Peyrot , W. J. , Schork , A. J. , Lee , S. H. & Wray , N. R . Genetic correlations of polygenic disease traits: from theory to practice . Nature Reviews Genetics 20 , 567 – 581 ( 2019 ). doi: 10.1038/s41576-019-0137-z OpenUrl CrossRef PubMed ↵ Sullivan , E. V. et al. Cognitive, emotion control, and motor performance of adolescents in the NCANDA study: Contributions from alcohol consumption, age, sex, ethnicity, and family history of addiction . Neuropsychology 30 , 449 – 473 ( 2016 ). doi: 10.1037/neu0000259 OpenUrl CrossRef PubMed ↵ The Brainstorm , C. , et al. Analysis of shared heritability in common disorders of the brain . Science 360 , eaap8757 ( 2018 ). doi: 10.1126/science.aap8757 OpenUrl Abstract / FREE Full Text ↵ Muennighoff , N. , Tazi , N. , Magne , L. & Reimers , N. in Conference of the European Chapter of the Association for Computational Linguistics . ↵ Nagel , M. , Watanabe , K. , Stringer , S. , Posthuma , D. & van der Sluis , S . Item-level analyses reveal genetic heterogeneity in neuroticism . Nature Communications 9 , 905 ( 2018 ). doi: 10.1038/s41467-018-03242-8 OpenUrl CrossRef PubMed ↵ Nagel , M. et al. Meta-analysis of genome-wide association studies for neuroticism in 449,484 individuals identifies novel genetic loci and pathways . Nat Genet 50 , 920 – 927 ( 2018 ). doi: 10.1038/s41588-018-0151-7 OpenUrl CrossRef PubMed ↵ Nagel , M. et al. Meta-analysis of genome-wide association studies for neuroticism in 449,484 individuals identifies novel genetic loci and pathways . Nat Genet 50 , 920 – 927 ( 2018 ). doi: 10.1038/s41588-018-0151-7 OpenUrl CrossRef PubMed ↵ Eysenck , S. B. , Eysenck , H. J. & Barrett , P . A revised version of the psychoticism scale . Personality and individual differences 6 , 21 – 29 ( 1985 ). OpenUrl ↵ Vaswani , A. et al. Attention is all you need . Advances in neural information processing systems 30 ( 2017 ). ↵ Devlin , J. , Chang , M.-W. , Lee , K. & Toutanova , K. in Proceedings of the 2019 conference of the North American chapter of the association for computational linguistics: human language technologies, volume 1 (long and short papers) . 4171 – 4186 . ↵ Radford , A. , Narasimhan , K. , Salimans , T. & Sutskever , I. ( ed OpenAI ) ( https://www.mikecaptain.com/resources/pdf/GPT-1.pdf , 2018 ). ↵ Radford , A. et al. ( ed OpenAI ) ( https://storage.prod.researchhub.com/uploads/papers/2020/06/01/language-models.pdf , 2019 ). ↵ Liu , Y. , et al. Roberta: A robustly optimized bert pretraining approach . arXiv preprint arXiv:1907.11692 ( 2019 ). ↵ Muennighoff , N. , Tazi , N. , Magne , L. & Reimers , N. 2014 – 2037 ( Association for Computational Linguistics ). Cao , H. Recent advances in text embedding: A Comprehensive Review of Top-Performing Methods on the MTEB Benchmark . ArXiv abs/2406.01607 ( 2024 ). Lee , J. , et al. Gecko: Versatile Text Embeddings Distilled from Large Language Models . ( 2024 ). OpenAI . ( https://openai.com/index/new-and-improved-embedding-model/ , 2023 ). OpenAI . ( https://openai.com/index/new-embedding-models-and-api-updates/ , 2024 ). View the discussion thread. Back to top Previous Next Posted March 27, 2025. Download PDF Data/Code Email Thank you for your interest in spreading the word about medRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following Encoding of pretrained large language models mirrors the genetic architectures of human psychological traits Message Subject (Your Name) has forwarded a page to you from medRxiv Message Body (Your Name) thought you would like to see this page from the medRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share Encoding of pretrained large language models mirrors the genetic architectures of human psychological traits Bohan Xu , Nick Obradovich , Wenjie Zheng , Robert Loughnan , Lucy Shao , Masaya Misaki , Wesley K. Thompson , Martin Paulus , Chun Chieh Fan medRxiv 2025.03.27.25324744; doi: https://doi.org/10.1101/2025.03.27.25324744 Share This Article: Copy Citation Tools Encoding of pretrained large language models mirrors the genetic architectures of human psychological traits Bohan Xu , Nick Obradovich , Wenjie Zheng , Robert Loughnan , Lucy Shao , Masaya Misaki , Wesley K. Thompson , Martin Paulus , Chun Chieh Fan medRxiv 2025.03.27.25324744; doi: https://doi.org/10.1101/2025.03.27.25324744 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Health Informatics Subject Areas All Articles Addiction Medicine (568) Allergy and Immunology (863) Anesthesia (300) Cardiovascular Medicine (4435) Dentistry and Oral Medicine (444) Dermatology (382) Emergency Medicine (608) Endocrinology (including Diabetes Mellitus and Metabolic Disease) (1509) Epidemiology (15227) Forensic Medicine (30) Gastroenterology (1124) Genetic and Genomic Medicine (6597) Geriatric Medicine (668) Health Economics (997) Health Informatics (4534) Health Policy (1368) Health Systems and Quality Improvement (1613) Hematology (540) HIV/AIDS (1264) Infectious Diseases (except HIV/AIDS) (15916) Intensive Care and Critical Care Medicine (1103) Medical Education (623) Medical Ethics (146) Nephrology (667) Neurology (6599) Nursing (346) Nutrition (998) Obstetrics and Gynecology (1144) Occupational and Environmental Health (957) Oncology (3332) Ophthalmology (974) Orthopedics (369) Otolaryngology (420) Pain Medicine (436) Palliative Medicine (130) Pathology (663) Pediatrics (1693) Pharmacology and Therapeutics (691) Primary Care Research (711) Psychiatry and Clinical Psychology (5447) Public and Global Health (9230) Radiology and Imaging (2198) Rehabilitation Medicine and Physical Therapy (1370) Respiratory Medicine (1196) Rheumatology (593) Sexual and Reproductive Health (712) Sports Medicine (530) Surgery (712) Toxicology (99) Transplantation (289) Urology (265) (function(){function c(){var b=a.contentDocument||a.contentWindow.document;if(b){var d=b.createElement('script');d.innerHTML="window.__CF$cv$params={r:'a002d84a8a01c13d',t:'MTc3OTUyNjU2MA=='};var a=document.createElement('script');a.src='/cdn-cgi/challenge-platform/scripts/jsd/main.js';document.getElementsByTagName('head')[0].appendChild(a);";b.getElementsByTagName('head')[0].appendChild(d)}}if(document.body){var a=document.createElement('iframe');a.height=1;a.width=1;a.style.position='absolute';a.style.top=0;a.style.left=0;a.style.border='none';a.style.visibility='hidden';document.body.appendChild(a);if('loading'!==document.readyState)c();else if(window.addEventListener)document.addEventListener('DOMContentLoaded',c);else{var e=document.onreadystatechange||function(){};document.onreadystatechange=function(b){e(b);'loading'!==document.readyState&&(document.onreadystatechange=e,c())}}}})();

Text is read by the "Ask this paper" AI Q&A widget below. Extraction quality varies by source — PMC NXML preserves structure cleanly, OA-HTML may include some navigation residue, and OA-PDF can have broken hyphenation. The publisher copy (via DOI) is the canonical version.

My notes (saved in your browser only)

Ask this paper AI returns verbatim quotes from the full text · source: preprint-html

Answers must be backed by verbatim quotes from this paper's full text. Hallucinated quotes are dropped automatically; if no verbatim passage answers the question, we say so. How this works

Citation neighborhood (no data yet)

We don't have any in-corpus citations linked to this paper yet. This is a recent paper (2025) — citers typically take a year or two to land, and the OpenAlex reference graph may still be filling in.

Source provenance

europepmc
last seen: 2026-05-20T01:45:00.602351+00:00
unpaywall
last seen: 2026-05-22T02:00:06.705733+00:00
License: CC-BY-4.0