Unsupervised Extractive Summarization of Psychedelic User Experience Reports

preprint OA: closed Public-Domain
📄 Open PDF Full text JSON View at publisher
AI-generated summary by claude@2026-07, 2026-07-16

This study developed and evaluated unsupervised extractive summarization methods, finding LexRank offered the best balance for psychedelic user reports, while SBERT captured more experiential depth at the cost of coherence.

One-sentence paraphrase of the abstract; not a substitute for reading it. No clinical advice. How this works

AI-generated deep summary by claude@2026-07, 2026-07-16 · read from full text

This study developed and evaluated an unsupervised extractive text summarization pipeline for psychedelic user experience reports, using 1,200 first-person narratives (400 each for LSD, psilocybin, and DMT) from Erowid. The authors trained and tuned models with a custom scoring function combining semantic coverage, narrative coherence, and an experiential-preservation metric, and compared three extractive approaches (LexRank, LSA with HDBSCAN clustering, and SBERT with Maximal Marginal Relevance) using GPT-4 as a calibrated rater with structured rubric scoring aggregated via TOPSIS. They found LexRank achieved the best overall balance, while SBERT produced higher content coverage and experiential depth but less coherence, with performance varying by substance due to differences in narrative structure and phenomenology. Key limitations explicitly include reliance on extractive methods, lack of reference summaries, and sensitivity to the scoring design, and the paper frames its contribution as a first step toward clinically usable summarization. The paper does not explicitly discuss endometriosis or adenomyosis; it was included in the corpus via a keyword match in the upstream search index.

Read from the paper's body, not the abstract. Not a substitute for reading the paper. No clinical advice. How this works

Abstract

A bstract Contemporary psychedelic research highlights the value of user experience reports, yet their verbose, subjective nature poses challenges for clinical utility. This is the first study to pioneer unsupervised automatic text summarization of psychedelic user experience reports, a domain where no human-annotated reference summaries exist. To address this gap, we developed a custom scoring function that integrates semantic coverage, narrative coherence, and a novel experiential preservation metric, enabling effective model training and hyperparameter tuning. We utilized three established extractive methods: LexRank, LSA with HDBSCAN clustering, and SBERT with Maximal Marginal Relevance, on 1,200 reports involving LSD, psilocybin, and DMT. Using GPT-4 as a calibrated rater under a structured rubric, supplemented by TOPSIS aggregation, results showed LexRank achieving the highest overall balance with SBERT excelling in content coverage and experiential depth but lagging in coherence. Our findings revealed trade-offs between content richness and narrative fluency, with performance varying across substance types due to differences in narrative structure and phenomenology. Limitations included reliance on extractive methods, lack of reference data, and sensitivity to scoring design. Future work should extend to abstractive methods, alternative weighting schemes, and expert adjudication to develop clinically usable summarization systems for psychedelic science.
Full text 43,068 characters · extracted from preprint-html · click to expand
Unsupervised Extractive Summarization of Psychedelic User Experience Reports | medRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-P4HH5NV'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search Unsupervised Extractive Summarization of Psychedelic User Experience Reports Shahidul Islam , View ORCID Profile Sakib Salam , Md Nahid Hasan doi: https://doi.org/10.1101/2025.08.22.25334176 Shahidul Islam 1 Department of Computer Science and Engineering, Fareast International University , Dhaka, Bangladesh Find this author on Google Scholar Find this author on PubMed Search for this author on this site Sakib Salam 2 Division of Biostatistics, Medical College of Wisconsin , Milwaukee, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Sakib Salam For correspondence: msalam{at}mcw.edu Md Nahid Hasan 3 Department of Mathematics, East Texas A&M University , Commerce, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site Abstract Full Text Info/History Metrics Data/Code Preview PDF A bstract Contemporary psychedelic research highlights the value of user experience reports, yet their verbose, subjective nature poses challenges for clinical utility. This is the first study to pioneer unsupervised automatic text summarization of psychedelic user experience reports, a domain where no human-annotated reference summaries exist. To address this gap, we developed a custom scoring function that integrates semantic coverage, narrative coherence, and a novel experiential preservation metric, enabling effective model training and hyperparameter tuning. We utilized three established extractive methods: LexRank, LSA with HDBSCAN clustering, and SBERT with Maximal Marginal Relevance, on 1,200 reports involving LSD, psilocybin, and DMT. Using GPT-4 as a calibrated rater under a structured rubric, supplemented by TOPSIS aggregation, results showed LexRank achieving the highest overall balance with SBERT excelling in content coverage and experiential depth but lagging in coherence. Our findings revealed trade-offs between content richness and narrative fluency, with performance varying across substance types due to differences in narrative structure and phenomenology. Limitations included reliance on extractive methods, lack of reference data, and sensitivity to scoring design. Future work should extend to abstractive methods, alternative weighting schemes, and expert adjudication to develop clinically usable summarization systems for psychedelic science. 1 Introduction 1.1 Background The renaissance of clinical research into psychedelics marks a significant paradigm shift in psychiatry and mental health care, characterized by a turbulent and dynamic timeline of scientific exploration and societal attitudes. Psychedelic compounds such as LSD, psilocybin, and mescaline were first investigated intensively in the 1950s and 1960s, when early clinical studies suggested promise for treating alcoholism, anxiety, and existential distress in terminal illness [ 1 , 2 ]. However, widespread abuse, growing political concern, and the U.S. Controlled Substances Act of 1970 led to an abrupt research moratorium [ 3 ]. After decades of dormancy, psychedelic science reemerged in the late 1990s and early 2000s, catalyzed by work at Johns Hopkins, Imperial College London, and other prominent research hubs. Modern trials employ rigorous methodologies, revealing that psilocybin can produce substantial and enduring reductions in depressive symptoms, even among treatment-resistant patients [ 4 ]. Likewise, MDMA and 5-MeO-DMT assisted therapy has yielded remarkable improvements in PTSD symptoms [ 5 , 6 ]. These recent findings have made psychedelics a key focus of modern psychiatric advancements breaking the barrier of pre-existing social norms.. Parallel to clinical trials, an expansive body of first-person trip narratives exists online via repositories like Erowid, Reddit, and PsychonautWiki. In contrast to formal clinical records, these user experience reports are often verbose narration filled with highly unstructured granular details, vivid emotional description, and subjective insights that reflect the complex, personal nature of psychedelic experiences. These seemingly story-like reports are actually invaluable for generating hypotheses, identifying risks, and understanding long-term effects of psychedelics. However, turning these detailed, unstructured narratives into concise, clinically relevant summaries utilizing Natural Language Processing techniques is quite challenging, which requires a balance of clarity, coherence, preservation as well as medical usefulness. 1.2 Prior Works and Research Gap Recent computational work has applied NLP to these narratives, but primarily for classification, linguistic profiling, and predictive modeling, sentiment analysis rather than structured summarization. For instance, Noah et al. conducted a large-scale study of 103 psychoactive substances using embedding-based models to identify variability in visual effects, such as movement, color, and patterns [ 7 ]. Their work demonstrated cross-substance distinctions but remained narrowly focused on perceptual categories. Al-Imam et al. analyzed over 2,100 reports of LSD and psilocybin, applying BERT, RoBERTa, and VADER methods to identify emotional polarity and introspective themes [ 8 ]. Sentiment analysis of this study showed that VADER produced more polarized results, while RoBERTa offered cautious and more accurate classifications. Lexicon analysis revealed that psilocybin (mushroom) reports often focused on introspection and altered time perception, while LSD reports emphasized cognitive disturbances. On the other hand, Biba and O’Shea analyzed a large corpus of reddit reports pertaining to the discussion of psychedelics using NLP infrastructure to find out public sentiment polarization towards the consumption [ 9 ]. Other studies highlight the potential of computational text analysis for understanding psychedelic experiences. Tagli-azucchi emphasized language as a “window into altered consciousness,” demonstrating semantic and structural markers of psychedelic states with predictive value for therapeutic outcomes [ 10 ]. Hase et al. extended this line of work by profiling distinct linguistic signatures across psychedelics and antidepressants, finding substance-specific language markers (e.g., MDMA as emotional, DMT as analytical) [ 11 ]. Furthermore, Cox et al. used topic modeling of 1,141 reports to predict substance reduction outcomes, showing clinical predictive utility but focusing on behavioral prediction rather than experiential condensation [ 12 ]. Nonetheless, these studies leave a significant gap in systematic summarization of full-length, phenomenology-rich trip reports. Numerous review works [ 13 – 15 ] divide the Automatic Text Summarization (ATS) techniques into two broad category: extractive and abstractive methods. Extractive summarization preserves original wording by selecting salient sentences, while abstractive approaches generate paraphrased digests. Clinical NLP has produced robust frameworks for summarizing medical records and patient notes, focusing on symptoms, diagnoses, and treatments. However, psychedelic trip reports differ fundamentally. They are subjective, nonlinear, and often metaphorical. Moreover, conventional evaluation frameworks often rely on reference summaries or objective clinical criteria, which are absent in this niche criterion. Therefore, the non-existence of effective ATS and reference summaries for psychedelic trip reports underscores the urgent need for innovative summarization methods tailored to these unique experiences. 1.3 Research Aim With this study, we aimed to address the challenges of summarizing psychedelic trip reports by developing a clinically oriented summarization pipeline that reduces narrative length by approximately 65-75% while preserving essential experiential and clinical content. We developed a custom scoring function for model training and hyperparameter tuning pertaining to this unique research in absence of gold-standard reference data. Additionally, we compared those extractive summarization models across narratives of LSD, psilocybin, and DMT, evaluating how their distinct phenomenological structures influence model performance. We believe that generated summaries from our research can reduce clinicians’ reading burden while retaining key experiential and risk-related information to support therapeutic applications. 2 Methodology 2.1 Data Collection and Preprocessing We collected 400 narrative reports each for DMT, psilocybin, and LSD (total 1,200 reports) from the Erowid experience archive, with permission from the copyright holders ( copyrights{at}erowid.com ). Previous research has demonstrated the value of Erowid as a corpus for computational and clinical investigations. For instance, Sanz et al. (2018) analyzed Erowid reports to identify symptom dimensions of hallucinogen use, while Swogger et al. (2015) conducted a qualitative thematic analysis of kratom user experiences hosted on Erowid.org, highlighting both therapeutic and adverse outcomes [ 16 , 17 ]. Following these precedents, we constructed a balanced psychedelic-specific dataset stratified by substance. We excluded reports shorter than 500 words or longer than 2,500 words to ensure narrative depth and analytical consistency. All reports were already anonymous and pseudonymized by Erowid moderators and users. Nonetheless, we conducted an additional named-entity recognition (NER) pass to remove any residual personal identifiers, ensuring full de-identification. To prepare the narratives for analysis, a systematic text cleaning and preprocessing pipeline was implemented. Structural metadata (e.g., “BODY WEIGHT,” “Dosage,” “Exp Year”) was truncated using regular expressions, retaining only the experiential content. Redundancy was reduced by eliminating duplicate sentences within each report. Text normalization proceeded in several stages: contractions were expanded (e.g., “don’t” to “do not”), clipped leading contractions (“’ve,” “’d”) were mapped to full grammatical forms (“i have,” “i would”), and parenthetical or bracketed insertions were removed. Punctuation was standardized by replacing em/en dashes and ellipses with commas, and collapsing repeated punctuation marks. Encoding artifacts (mojibake) were corrected via rule-based replacement of malformed Unicode sequences (e.g., “’” to “’”). We further applied case normalization, split camel-Case expressions, and enforced consistent whitespace. In addition to text cleaning, demographic metadata were extracted where available. Reported ages were parsed from the “Age at time of experience” field, and weights expressed in pounds or stone were converted to kilograms for consistency. Finally, we conducted descriptive analyses of demographic and textual distributions across substances. We visualized gender proportions with pie charts and depicted age and weight distributions with histograms and strip plots. Moreover, we assessed narrative length variations via word-count histograms. These preprocessing steps ensured the dataset was structurally consistent, demographically harmonized, and analytically robust for subsequent model training, parameter tuning, and evaluation. In our workflow, we divided the corpus into ~75% for model training, hyperparameter tuning and ~25% for evaluation, stratified uniformly across LSD, psilocybin, and DMT to preserve substance-specific balance. We trained and tuned the models on the ~75% subset using an Optuna-based parameter search (Akiba et al., 2019), and then applied the optimal configurations to the held-out ~25% subset to generate summaries. These summaries were subsequently evaluated using the LLM rubric framework described afterwards. This framework leverages large language models as evaluators, a scalable alternative to human judgment in NLP task [ 18 , 19 ]. 2.2 Models and Hyperparameter Tuning We evaluated three unsupervised extractive summarization approaches. The first was LexRank, a graph-based centrality algorithm. LexRank represents sentences as nodes in a similarity graph, with edges weighted by TF–IDF cosine similarity [ 20 ]. Low-similarity edges are pruned, and sentence salience is computed through eigenvector centrality. The top-ranked sentences are finally selected according to a target compression ratio. The second approach combined Latent Semantic Analysis (LSA) with HDBSCAN clustering to identify semantically coherent sentence groups. LSA reduces TF–IDF sentence vectors into a lower-dimensional latent space using singular value decomposition (SVD) [ 21 ]. Sentences are subsequently grouped using HDBSCAN, a hierarchical density-based clustering algorithm [ 22 ]. This hybrid approach balances topical representativeness with phenomenological richness. The third approach employed Sentence-BERT (SBERT) with Maximal Marginal Relevance (MMR) for embedding-based sentence selection. SBERT generates dense semantic embeddings using transformer encoders [ 23 ]. MMR iteratively selects sentences by balancing document relevance and diversity [ 24 ]. Relevance is computed against a document embedding, while diversity penalizes redundancy among selected sentences. Additional tuning adjusts position bias and similarity thresholds. This framework enforces both semantic fidelity and experiential diversity, making it particularly well-suited for the multi-faceted nature of psychedelic narratives. All three models were tuned using Optuna with a Tree-structured Parzen Estimator (TPE) sampler and Median Pruner, allowing efficient exploration of hyperparameter spaces and early stopping of unpromising trials. The optimization objective was a custom composite score balancing semantic fidelity, phenomenological preservation, and narrative coherence, as expressed by: We computed semantic similarity as cosine similarity between the summary and source (TF–IDF for LexRank and LSA+HDBSCAN; SBERT embeddings for SBERT+MMR). To evaluate experiential preservation, we curated a lexicon of terms spanning six distinct domains, obtaining initial vocabulary with prior psychometric work on altered states of consciousness by Studerus et al.[ 25 ]. Then we expanded it using GPT-4 to capture synonyms and phrase variants. We measured preservation as the proportion of experiential terms in the source that appeared in the summary. To assess sentence coherence, we averaged pairwise cosine similarity between consecutive summary sentences. A representative excerpt of our experiential lexicon with respective domains is shown in Table 1 , with the complete list provided in Appendix A. Furthermore, the hyperparameter search space of every model is summarized in Table 2 . View this table: View inline View popup Download powerpoint Table 1: Representative Experiential Terms by Category. The full lexicon is provided in Appendix A. View this table: View inline View popup Download powerpoint Table 2: Hyperparameter Configuration and Tuning Ranges for Each Model. 2.3 LLM Evaluation We evaluated the extracted summaries with optimal hyper-parameters using GPT-4 acting as a calibrated expert rater under a structured rubric. The exact prompt given in Table 3 is reproduced to ensure transparency and reproducibility. View this table: View inline View popup Download powerpoint Table 3: LLM Evaluation Prompt (GPT-4) used for psychedelic trip report summarization. 3 Results Figure 1 illustrates the distribution of word counts and demographic attributes across psilocybin, LSD, and DMT reports. Word counts varied substantially across substances. LSD and psilocybin reports showed relatively normal distributions with peaks around 1,200–1,500 words, whereas DMT reports tended to be shorter, concentrated mainly between 500–1,000 words. Gender distribution was consistently male-dominated (79.5% for psilocybin, 76.2% for LSD, and 83.2% for DMT), with female contributions comprising less than one-fifth of the dataset. The age profiles revealed that the majority of participants fell within the 15–30 age range, with a small fraction of users showing slightly more spread into older cohorts. Weight distributions were broadly consistent across substances, although LSD users showed least outliers within the weight range. Download figure Open in new tab Figure 1. Distribution of word counts and demographics across substances (psilocybin, LSD, DMT). Table 4 summarizes the final hyperparameter configurations that yielded the best-performing models, along with their corresponding average custom scores. Table 5 presents the GPT-4 evaluation scores (1–5 scale) for each model across the five criteria, with averaged Overall Scores reported per substance. Furthermore, Table 5 reports GPT-4 rubric scores (1–5) per substance and model. View this table: View inline View popup Download powerpoint Table 4: Optimal hyperparameter settings with average Custom Score on test data. View this table: View inline View popup Download powerpoint Table 5: GPT-4 evaluation scores across criteria (1–5) and Overall. Figure 2 presents radar chart visualizations, comparing the relative strengths and weaknesses of each model across evaluation dimensions by substance. SBERT showed high content coverage and experiential preservation but lower coherence, while LexRank maintained moderate scores across dimensions, and LSA excelled primarily in compression quality. Table 6 reports the TOPSIS closeness coefficients, synthesizing the multi-criteria evaluation scores into a ranked perspective across substances as well as overall. The results indicate that LexRank achieved the highest overall score, followed by SBERT and then LSA. View this table: View inline View popup Download powerpoint Table 6: TOPSIS closeness coefficients (0–1) across models and substances under equal weights. View this table: View inline View popup Table 7: Full experiential lexicon categorized into emotional, sensory, cognitive, physical, mystical, and temporal domains. Download figure Open in new tab Figure 2. Radar chart visualization of model performance by substance across five criteria. 4 Discussion Across substances, the three extractive pipelines showed complementary strengths that mirror well-documented trade-offs in clinical summarization. SBERT generally achieved higher LLM-based scores on content coverage, clinical relevance, and experiential preservation, though this often came at the expense of narrative coherence and length control, leading to longer or less organized summaries. LexRank and LSA, in contrast, tended to produce shorter and more fluent outputs with stronger coherence, but more frequently omitted dosing details, time course, or nuanced phenomenological content. This pattern echoes findings from medical text summarization research where content richness and readability often pull in opposite directions [ 26 , 27 ]. When we aggregated the five evaluation dimensions with equal weights using TOPSIS, the results emphasized balance rather than maximal content capture. LexRank achieved the highest overall closeness score (0.54), followed by SBERT (0.51) and LSA (0.49). These differences highlight how modest deficits in coherence and compression can lower composite rankings even when content retention and clinical relevance are strong. Importantly, the rankings produced by the LLM rubric and TOPSIS should be seen as complementary: the former reveals performance on individual criteria, while the latter captures holistic balance across them [ 28 ]. Substance-specific patterns also suggest that no single method consistently dominates. SBERT performed better on dense and phenomenology-rich DMT narratives, where its embeddings captured subjective and temporal detail that rule-based methods tended to miss. LexRank was comparatively stronger on psilocybin reports, where recurring thematic motifs and structural coherence favored sentence-graph centrality. LSD reports, often shorter and less structurally complex, narrowed the gap between methods and allowed LSA to perform adequately despite its lower experiential retention. These variations indicate that model choice may depend as much on the structure and phenomenology of the source material as on overall algorithmic design. From a practical standpoint, SBERT appears more suitable when completeness and clinical traceability are prioritized, such as in safety reviews or research coding. However, a lightweight post-processing stage that enforces target length, removes redundancies, and improves ordering could address its coherence limitations. LexRank is more appropriate for dashboards or registries where brevity and readability are essential, and it may also serve as a trimming layer for embedding-based methods. LSA remains a viable baseline when brevity and coherence are emphasized, although it risks losing important subjective detail in longer or more phenomenological narratives. These recommendations align with clinical NLP findings in radiology, discharge summary generation, and patient information summarization, where extraction and abstraction are often combined to balance factuality, readability, and coverage [ 29 , 30 ]. Finally, the evaluation framework itself warrants reflection. Using GPT-4 as a calibrated rater under a fixed rubric proved useful for distinguishing between models. However, like other LLM-as-judge settings, it remains sensitive to prompt design and length effects [ 31 ]. Complementing this approach with TOPSIS helped make explicit the assumptions behind weighting choices and showed how rankings can shift depending on the balance of dimensions considered. Our study also has several limitations that should be noted. The evaluation was based on a relatively small sample of reports, which restricts generalizability. We focused exclusively on extractive pipelines and did not test abstractive summarization methods, which may offer different advantages and challenges. The absence of a standardized reference dataset limited our ability to benchmark against prior work. Our scoring function and weighting scheme, while transparent, represent only one possible configuration, and alternative formulations may yield different outcomes. Finally, relying on GPT-4 as a rater raises questions of reproducibility and potential bias. Despite these constraints, the findings suggest clear directions for future work. Larger and more diverse collections of psychedelic reports would allow for more robust evaluation. Human adjudication could be incorporated alongside LLM judgments to strengthen reliability. Systematic testing of alternative scoring functions and weighting schemes would clarify how evaluation choices shape rankings. Abstractive and hybrid extraction–abstraction models deserve further exploration, as they may better capture the balance between factual coverage and narrative coherence. Taken together, these directions can help advance toward summarization systems that are both scientifically rigorous and clinically useful. Data Availability We collected 400 narrative reports each for DMT, psilocybin, and LSD (total 1,200 reports) from the Erowid experience archive, with permission from the copyright holders ( copyrights{at}erowid.com ). https://www.erowid.org/ Data Availability The full-text reports are hosted by Erowid and cannot be redistributed due to copyright restrictions. Preprocessing code, analysis scripts, results and a representative example of generated summaries are available at: https://github.com/Shahidul2/Psychedelics-summary . Author Contributions S.I. conceived the study, analyzed data, and interpreted results. S.S. collected data and conducted the literature review. M.N.H. supervised the study and revised the manuscript. All authors contributed equally to writing. Ethics Statement This research complies with ethical publication standards and institutional review requirements. The study did not involve human subjects research, animal experimentation, or the collection of sensitive personal data. All source materials consisted of publicly available, fully anonymized narrative reports from Erowid, accessed with permission from the site administrators. A NER pipeline removed any residual personal identifiers to ensure total de-identification. Disclosure Statement The authors have no conflicts of interest and have not received any external funding. A Full Experiential Lexicon Footnotes badhon279{at}gmail.com msalam{at}mcw.edu Nahid.Hasan{at}etamu.edu References [1]. ↵ Mark A. Geyer . “ A Brief Historical Overview of Psychedelic Research ”. In: Biological Psychiatry: Cognitive Neuroscience and Neuroimaging 9 . 5 ( May 2024 ), pp. 464 – 471 . ISSN: 24519022 . DOI: 10.1016/j.bpsc.2023.11.003 . URL: https://linkinghub.elsevier.com/retrieve/pii/S2451902223003142 (visited on 08/18/2025 ). OpenUrl CrossRef [2]. ↵ David E. Nichols and Hannes Walter . “ The History of Psychedelics in Psychiatry ”. In: Pharmacopsychiatry 54 . 4 ( July 2021 ), pp. 151 – 166 . ISSN: 0176-3679 , 1439-0795. DOI: 10.1055/a-1310-3990 . URL: http://www.thieme-connect.de/DOI/DOI?10.1055/a-1310-3990 (visited on 08/18/2025 ). OpenUrl CrossRef PubMed [3]. ↵ Sean J. Belouin and Jack E. Henningfield . “ Psychedelics: Where we are now, why we got here, what we must do ”. In: Neuropharmacology 142 ( Nov . 2018 ), pp. 7 – 19 . ISSN: 00283908 . DOI: 10.1016/j.neuropharm.2018.02.018 . URL: https://linkinghub.elsevier.com/retrieve/pii/S0028390818300753 (visited on 08/18/2025 ). OpenUrl CrossRef PubMed [4]. ↵ Robin L Carhart-Harris et al. “ Psilocybin with psychological support for treatment-resistant depression: an open-label feasibility study ”. In: The Lancet Psychiatry 3 . 7 ( July 2016 ), pp. 619 – 627 . ISSN: 22150366 . DOI: 10.1016/S2215-0366(16)30065-7 . URL: https://linkinghub.elsevier.com/retrieve/pii/S2215036616300657 (visited on 08/18/2025 ). OpenUrl CrossRef PubMed [5]. ↵ Jennifer M. Mitchell et al. “ MDMA-assisted therapy for moderate to severe PTSD: a randomized, placebo-controlled phase 3 trial ”. In: Nature Medicine 29 . 10 ( Oct . 2023 ), pp. 2473 – 2480 . ISSN: 1078-8956 , 1546-170X. DOI: 10.1038/s41591-023-02565-4 . URL: https://www.nature.com/articles/s41591-023-02565-4 (visited on 08/18/2025 ). OpenUrl CrossRef PubMed [6]. ↵ Alan Kooi Davis et al. “ Open-label study of consecutive ibogaine and 5-MeO-DMT assisted-therapy for trauma-exposed male Special Operations Forces Veterans: prospective data from a clinical program in Mexico ”. In: The American Journal of Drug and Alcohol Abuse 49 . 5 ( Sept . 3, 2023 ), pp. 587 – 596 . ISSN: 0095-2990 , 1097-9891. DOI: 10.1080/00952990.2023.2220874 . URL: https://www.tandfonline.com/doi/full/10.1080/00952990.2023.2220874 (visited on 08/18/2025 ). OpenUrl CrossRef PubMed [7]. ↵ Sean Noah et al. “ A novel method for quantitative analysis of subjective experience reports: application to psychedelic visual experiences ”. In: Frontiers in Psychology 15 ( Dec . 6, 2024 ), p. 1397064 . ISSN: 1664-1078 . DOI: 10.3389/fpsyg.2024.1397064 . URL: https://www.frontiersin.org/articles/10.3389/fpsyg.2024.1397064/full (visited on 08/18/2025 ). OpenUrl CrossRef [8]. ↵ Ahmed Al-Imam et al. “ Opinion Mining of Erowid’s Experience Reports on LSD and Psilocybin-Containing Mushrooms ”. In: Drug Safety 48 . 5 ( May 2025 ), pp. 559 – 575 . ISSN: 0114-5916 , 1179-1942. DOI: 10.1007/s40264-025-01530-z . URL: https://link.springer.com/10.1007/s40264-025-01530-z (visited on 08/18/2025 ). OpenUrl CrossRef [9]. ↵ Brandon Biba and Brian A. O’Shea . “ Exploring Public Sentiments of Psychedelics Versus Other Substances: A Reddit-Based Natural Language Processing Study ”. In: Journal of Psychoactive Drugs ( May 30 , 2025 ), pp. 1 – 11 . ISSN: 0279-1072 , 2159-9777. DOI: 10.1080/02791072.2025.2511750 . URL: https://www.tandfonline.com/doi/full/10.1080/02791072.2025.2511750 (visited on 08/18/2025 ). OpenUrl CrossRef [10]. ↵ Enzo Tagliazucchi . “ Language as a Window Into the Altered State of Consciousness Elicited by Psychedelic Drugs ”. In: Frontiers in Pharmacology 13 ( Mar . 22, 2022 ), p. 812227 . ISSN: 1663-9812 . DOI: 10.3389/fphar.2022.812227 . URL: https://www.frontiersin.org/articles/10.3389/fphar.2022.812227/full (visited on 08/18/2025 ). OpenUrl CrossRef [11]. ↵ Adrian Hase et al. “ Analysis of recreational psychedelic substance use experiences classified by substance ”. In: Psychopharmacology 239 . 2 ( Feb . 2022 ), pp. 643 – 659 . ISSN: 0033-3158 , 1432-2072. DOI: 10.1007/s00213-022-06062-3 . URL: https://link.springer.com/10.1007/s00213-022-06062-3 (visited on 08/18/2025 ). OpenUrl CrossRef PubMed [12]. ↵ David J. Cox , Albert Garcia-Romeu , and Matthew W. Johnson . “ Predicting changes in substance use following psychedelic experiences: natural language processing of psychedelic session narratives ”. In: The American Journal of Drug and Alcohol Abuse 47 . 4 ( July 4 , 2021 ), pp. 444 – 454 . ISSN: 0095-2990 , 1097-9891. DOI: 10.1080/00952990.2021.1910830 . URL: https://www.tandfonline.com/doi/full/10.1080/00952990.2021.1910830 (visited on 08/18/2025 ). OpenUrl CrossRef PubMed [13]. ↵ Nouf Ibrahim Altmami and Mohamed El Bachir Menai . “ Automatic summarization of scientific articles: A survey ”. In: Journal of King Saud University - Computer and Information Sciences 34 . 4 ( Apr . 2022 ), pp. 1011 – 1028 . ISSN: 13191578 . DOI: 10.1016/j.jksuci.2020.04.020 . URL: https://linkinghub.elsevier.com/retrieve/pii/S1319157820303554 (visited on 08/16/2025 ). OpenUrl CrossRef [14]. Bilal Khan et al. “ Exploring the Landscape of Automatic Text Summarization: A Comprehensive Survey ”. In: IEEE Access 11 ( 2023 ), pp. 109819 – 109840 . ISSN: 2169-3536 . DOI: 10.1109/ACCESS.2023.3322188 . URL: https://ieeexplore.ieee.org/document/10272614/ (visited on 08/16/2025 ). OpenUrl CrossRef [15]. ↵ Avaneesh Kumar Yadav et al. “ State-of-the-art approach to extractive text summarization: a comprehensive review ”. In: Multimedia Tools and Applications 82 . 19 ( Aug . 1, 2023 ), pp. 29135 – 29197 . ISSN: 1573-7721 . DOI: 10.1007/s11042-023-14613-9 . URL: https://doi.org/10.1007/s11042-023-14613-9 . OpenUrl CrossRef [16]. ↵ Camila Sanz et al. “ The Experience Elicited by Hallucinogens Presents the Highest Similarity to Dreaming within a Large Database of Psychoactive Substance Reports ”. In: Frontiers in Neuroscience 12 ( Jan . 22, 2018 ), p. 7 . ISSN: 1662-453 X. DOI: 10.3389/fnins.2018.00007 . URL: http://journal.frontiersin.org/article/10.3389/fnins.2018.00007/full (visited on 08/19/2025 ). OpenUrl CrossRef [17]. ↵ Marc T. Swogger et al. “ Experiences of Kratom Users: A Qualitative Analysis ”. In: Journal of Psychoactive Drugs 47 . 5 ( Oct . 20, 2015 ), pp. 360 – 367 . ISSN: 0279-1072 , 2159-9777. DOI: 10.1080/02791072.2015.1096434 . URL: http://www.tandfonline.com/doi/full/10.1080/02791072.2015.1096434 (visited on 08/19/2025 ). OpenUrl CrossRef [18]. ↵ Anna Bavaresco et al . LLMs instead of Human Judges? A Large Scale Empirical Study across 20 NLP Evaluation Tasks . Version Number: 3. 2024 . DOI: 10.48550/ARXIV.2406.18403 . URL: https://arxiv.org/abs/2406.18403 (visited on 08/19/2025 ). OpenUrl CrossRef [19]. ↵ Nitay Calderon , Roi Reichart , and Rotem Dror . The Alternative Annotator Test for LLM-as-a-Judge: How to Statistically Justify Replacing Human Annotators with LLMs . Version Number: 4. 2025 . DOI: 10.48550/ARXIV.2501.10970 . URL: https://arxiv.org/abs/2501.10970 (visited on 08/19/2025 ). OpenUrl CrossRef [20]. ↵ G. Erkan and D. R. Radev . “ LexRank: Graph-based Lexical Centrality as Salience in Text Summarization ”. In: Journal of Artificial Intelligence Research 22 ( Dec . 1, 2004 ), pp. 457 – 479 . ISSN: 1076-9757 . DOI: 10.1613/jair.1523 . URL: https://jair.org/index.php/jair/article/view/10396 (visited on 08/19/2025 ). OpenUrl CrossRef [21]. ↵ Qianqian Qi et al. “ A comparison of latent semantic analysis and correspondence analysis of document-term matrices ”. In: Natural Language Engineering 30 . 4 ( July 2024 ), pp. 722 – 752 . ISSN: 1351-3249 , 1469-8110. DOI: 10.1017/S1351324923000244 . URL: https://www.cambridge.org/core/product/identifier/S1351324923000244/type/journal_article (visited on 08/19/2025 ). OpenUrl CrossRef [22]. ↵ Leland McInnes , John Healy , and Steve Astels . “ hdbscan: Hierarchical density based clustering ”. In: The Journal of Open Source Software 2 . 11 ( Mar . 21, 2017 ), p. 205 . ISSN: 2475-9066 . DOI: 10.21105/joss.00205 . URL: http://joss.theoj.org/papers/10.21105/joss.00205 (visited on 08/19/2025 ). OpenUrl CrossRef [23]. ↵ Amol P. Bhopale and Ashish Tiwari . “ Transformer based contextual text representation framework for intelligent information retrieval ”. In: Expert Systems with Applications 238 ( Mar . 2024 ), p. 121629 . ISSN: 09574174 . DOI: 10.1016/j.eswa.2023.121629 . URL: https://linkinghub.elsevier.com/retrieve/pii/S0957417423021310 (visited on 08/19/2025 ). OpenUrl CrossRef [24]. ↵ Salima Lamsiyah et al. “ Unsupervised query-focused multi-document summarization based on transfer learning from sentence embedding models, BM25 model, and maximal marginal relevance criterion ”. In: Journal of Ambient Intelligence and Humanized Computing 14 . 3 ( Mar . 2023 ), pp. 1401 – 1418 . ISSN: 1868-5137 , 1868-5145. DOI: 10.1007/s12652-021-03165-1 . URL: https://link.springer.com/10.1007/s12652-021-03165-1 (visited on 08/19/2025 ). OpenUrl CrossRef [25]. ↵ Vaughan Bell Erich Studerus , Alex Gamma , and Franz X. Vollenweider . “ Psychometric Evaluation of the Altered States of Consciousness Rating Scale (OAV) ”. In: PLoS ONE 5 . 8 ( Aug . 31, 2010 ). Ed. by Vaughan Bell , e12412 . ISSN: 1932-6203 . DOI: 10.1371/journal.pone.0012412 . URL: https://dx.plos.org/10.1371/journal.pone.0012412 (visited on 08/19/2025 ). OpenUrl CrossRef PubMed [26]. ↵ Nensi Bralić et al. “ Conclusiveness, readability and textual characteristics of plain language summaries from medical and non-medical organizations: a cross-sectional study ”. In: Scientific Reports 14 . 1 ( Mar . 12, 2024 ), p. 6016 . ISSN: 2045-2322 . DOI: 10.1038/s41598-024-56727-6 . URL: https://www.nature.com/articles/s41598-024-56727-6 (visited on 08/19/2025 ). OpenUrl CrossRef [27]. ↵ Bryce Picton et al. “ Assessing AI Simplification of Medical Texts: Readability and Content Fidelity ”. In: International Journal of Medical Informatics 195 ( Mar . 2025 ), p. 105743 . ISSN: 13865056 . DOI: 10.1016/j.ijmedinf.2024.105743 . URL: https://linkinghub.elsevier.com/retrieve/pii/S1386505624004064 (visited on 08/19/2025 ). OpenUrl CrossRef [28]. ↵ Majid Behzadian et al. “ A state-of the-art survey of TOPSIS applications ”. In: Expert Systems with Applications 39 . 17 ( Dec . 2012 ), pp. 13051 – 13069 . ISSN: 09574174 . DOI: 10.1016/j.eswa.2012.05.056 . URL: https://linkinghub.elsevier.com/retrieve/pii/S0957417412007725 (visited on 08/19/2025 ). OpenUrl CrossRef [29]. ↵ Daniel Keszthelyi et al. “ Patient Information Summarization in Clinical Settings: Scoping Review ”. In: JMIR Medical Informatics 11 ( Nov . 28, 2023 ), e44639 . ISSN: 2291-9694 . DOI: 10.2196/44639 . URL: https://medinform.jmir.org/2023/1/e44639 (visited on 08/19/2025 ). OpenUrl CrossRef [30]. ↵ Yuhao Zhang et al. “ Optimizing the Factual Correctness of a Summary: A Study of Summarizing Radiology Reports ”. In: Proceedings of the 58th Annual Meeting of the Association for Computational Linguistics. Proceedings of the 58th Annual Meeting of the Association for Computational Linguistics. Online: Association for Computational Linguistics , 2020 , pp. 5108 – 5120 . DOI: 10.18653/v1/2020.acl-main.458 . URL: https://www.aclweb.org/anthology/2020.acl-main.458 (visited on 08/19/2025 ). OpenUrl CrossRef [31]. ↵ Yupeng Chang et al. “ A Survey on Evaluation of Large Language Models ”. In: ACM Transactions on Intelligent Systems and Technology 15 . 3 ( June 30 , 2024 ), pp. 1 – 45 . ISSN: 2157-6904 , 2157-6912. DOI: 10.1145/3641289 . URL: https://dl.acm.org/doi/10.1145/3641289 (visited on 08/19/2025 ). OpenUrl CrossRef View the discussion thread. Back to top Previous Next Posted August 27, 2025. Download PDF Data/Code Email Thank you for your interest in spreading the word about medRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following Unsupervised Extractive Summarization of Psychedelic User Experience Reports Message Subject (Your Name) has forwarded a page to you from medRxiv Message Body (Your Name) thought you would like to see this page from the medRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share Unsupervised Extractive Summarization of Psychedelic User Experience Reports Shahidul Islam , Sakib Salam , Md Nahid Hasan medRxiv 2025.08.22.25334176; doi: https://doi.org/10.1101/2025.08.22.25334176 Share This Article: Copy Citation Tools Unsupervised Extractive Summarization of Psychedelic User Experience Reports Shahidul Islam , Sakib Salam , Md Nahid Hasan medRxiv 2025.08.22.25334176; doi: https://doi.org/10.1101/2025.08.22.25334176 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Health Informatics Subject Areas All Articles Addiction Medicine (568) Allergy and Immunology (863) Anesthesia (300) Cardiovascular Medicine (4435) Dentistry and Oral Medicine (444) Dermatology (382) Emergency Medicine (608) Endocrinology (including Diabetes Mellitus and Metabolic Disease) (1509) Epidemiology (15229) Forensic Medicine (30) Gastroenterology (1124) Genetic and Genomic Medicine (6600) Geriatric Medicine (668) Health Economics (997) Health Informatics (4536) Health Policy (1368) Health Systems and Quality Improvement (1613) Hematology (541) HIV/AIDS (1264) Infectious Diseases (except HIV/AIDS) (15916) Intensive Care and Critical Care Medicine (1103) Medical Education (623) Medical Ethics (146) Nephrology (667) Neurology (6599) Nursing (346) Nutrition (998) Obstetrics and Gynecology (1144) Occupational and Environmental Health (957) Oncology (3332) Ophthalmology (974) Orthopedics (369) Otolaryngology (420) Pain Medicine (436) Palliative Medicine (130) Pathology (663) Pediatrics (1693) Pharmacology and Therapeutics (691) Primary Care Research (711) Psychiatry and Clinical Psychology (5447) Public and Global Health (9232) Radiology and Imaging (2198) Rehabilitation Medicine and Physical Therapy (1370) Respiratory Medicine (1196) Rheumatology (593) Sexual and Reproductive Health (712) Sports Medicine (530) Surgery (712) Toxicology (99) Transplantation (289) Urology (265) (function(){function c(){var b=a.contentDocument||a.contentWindow.document;if(b){var d=b.createElement('script');d.innerHTML="window.__CF$cv$params={r:'a008d6db9875c13d',t:'MTc3OTU4OTQxNg=='};var a=document.createElement('script');a.src='/cdn-cgi/challenge-platform/scripts/jsd/main.js';document.getElementsByTagName('head')[0].appendChild(a);";b.getElementsByTagName('head')[0].appendChild(d)}}if(document.body){var a=document.createElement('iframe');a.height=1;a.width=1;a.style.position='absolute';a.style.top=0;a.style.left=0;a.style.border='none';a.style.visibility='hidden';document.body.appendChild(a);if('loading'!==document.readyState)c();else if(window.addEventListener)document.addEventListener('DOMContentLoaded',c);else{var e=document.onreadystatechange||function(){};document.onreadystatechange=function(b){e(b);'loading'!==document.readyState&&(document.onreadystatechange=e,c())}}}})();

Text is read by the "Ask this paper" AI Q&A widget below. Extraction quality varies by source — PMC NXML preserves structure cleanly, OA-HTML may include some navigation residue, and OA-PDF can have broken hyphenation. The publisher copy (via DOI) is the canonical version.

My notes (saved in your browser only)

Ask this paper AI returns verbatim quotes from the full text · source: preprint-html

Answers must be backed by verbatim quotes from this paper's full text. Hallucinated quotes are dropped automatically; if no verbatim passage answers the question, we say so. How this works

Citation neighborhood (no data yet)

We don't have any in-corpus citations linked to this paper yet. This is a recent paper (2025) — citers typically take a year or two to land, and the OpenAlex reference graph may still be filling in.

Source provenance

europepmc
last seen: 2026-05-20T01:45:00.602351+00:00
unpaywall
last seen: 2026-05-26T02:00:01.498150+00:00
License: Public-Domain