Improve data management in register-based research: Transition from CSV to Parquet

preprint OA: closed
📄 Open PDF Full text JSON View at publisher

Abstract

Aims To identify an efficient file format for data delivery from large administrative registers for research, testing the full workflow from data extraction to research data management. Methods Through collaboration between a data delivery and a research department within the same institute, we evaluated each step from data extraction and delivery to management and usage, comparing CSV and Parquet formats. Results Switching from the unstructured, text-based CSV format to the highly structured Parquet format significantly optimized all processes by reducing file sizes and saving processing time. The Parquet format also provided access to advanced data management techniques, simplifying further work. Despite these advantages, the basic programming required for Parquet format is not very different from that for CSV. We provide a tutorial and examples as online supplement. Conclusions We strongly recommend replacing CSV files with contemporary data formats. The Parquet file format proved to be an excellent option throughout the entire process from data extraction to research implementation
Full text 30,828 characters · extracted from preprint-html · click to expand
Improve data management in register-based research: Transition from CSV to Parquet | medRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-P4HH5NV'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search Improve data management in register-based research: Transition from CSV to Parquet View ORCID Profile Simone Rahel Fenk , View ORCID Profile Kari Furu , View ORCID Profile Inger Johanne Bakken doi: https://doi.org/10.1101/2025.10.15.25337992 Simone Rahel Fenk 1 Helseplattformen , Trondheim, Norway 4 Department of Data Distribution, Patient Registries, Norwegian Institute of Public Health , Trondheim, Norway Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Simone Rahel Fenk Kari Furu 2 Department of Chronic Diseases, Norwegian Institute of Public Health , Oslo, Norway 3 Centre for Fertility and Health, Norwegian Institute of Public Health , Oslo, Norway Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Kari Furu Inger Johanne Bakken 2 Department of Chronic Diseases, Norwegian Institute of Public Health , Oslo, Norway Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Inger Johanne Bakken For correspondence: inger.johanne.bakken{at}fhi.no Abstract Full Text Info/History Metrics Data/Code Preview PDF Abstract Aims To identify an efficient file format for data delivery from large administrative registers for research, testing the full workflow from data extraction to research data management. Methods Through collaboration between a data delivery and a research department within the same institute, we evaluated each step from data extraction and delivery to management and usage, comparing CSV and Parquet formats. Results Switching from the unstructured, text-based CSV format to the highly structured Parquet format significantly optimized all processes by reducing file sizes and saving processing time. The Parquet format also provided access to advanced data management techniques, simplifying further work. Despite these advantages, the basic programming required for Parquet format is not very different from that for CSV. We provide a tutorial and examples as online supplement. Conclusions We strongly recommend replacing CSV files with contemporary data formats. The Parquet file format proved to be an excellent option throughout the entire process from data extraction to research implementation Introduction The Nordic countries are known for their population-based health registers. By means of the personal identity number (PIN) unique to every citizen, registers can be linked to each other and to other data sources, such as census data and information on education, income, and family structure. 1 The medical birth registers and the cause-of-death registers contain information on one-time events in a person’s life. Other registers, like the medical quality registers, are usually dedicated to conditions or treatments that are relatively rare. Even though such registers are population-based, size is not usually an issue in research when it comes to data management or analysis. The administrative health registers, however, contain broad information on all health care utilization in the country and rapidly accumulate large amount of data over time. For example, the Norwegian Patient Registry (NPR) and the Norwegian Registry of Primary Health Care (NRPHC) contain information on all inpatient and outpatient encounters in specialist health care from 2008 and in primary health care from 2016, respectively. 2 The Norwegian Prescribed Registry (NorPD) holds data on all prescription fills at Norwegian pharmacies from 2004 onwards. 3 For context, during a typical month in 2024, there were approximately 1.4 million general practitioner’s consultations (recorded in NRPHC) and 0.8 million outpatient visits at somatic hospitals (recorded in NPR). 4 , 5 These data represent only a small fraction of what is reported to NRPHC and NPR. Although NorPD does not present online statistics on the total annual number of prescriptions filled, it reports that more than 4 million people (75.3 % of the population) filled at least one prescription in 2024, with defined daily doses surpassing three billion. 6 Data in modern, well-functioning registers are highly organized and well-structured, with substantial human expertise and advanced technological resources dedicated to maintaining up-to-date scalable registers. The possibilities and advantages of using data from the mandatory, population-based health registers in the Nordic countries for research purposes are well-documented. 7 , 8 Studies often apply advanced statistical methods within carefully planned study designs, such as the increasingly popular target trial emulation approach. 9 These advanced and modern efforts differ notably from the workflow when data are extracted from databases and delivered to researchers. In our experience, the data format for transfer from registry to research project has not been changed since the early days of research using data from administrative registries. In this paper, we demonstrate how transitioning from a traditional unstructured row-based data file format to a contemporary structured column-oriented data file format significantly enhances data storage efficiency and improves data management capabilities. Our case is a large research project on pregnancy and medication use, Drugs in Childhood and Reproductive Life with second author (KF) as a project leader. 10 , 11 Materials and Methods In this collaboration between a data delivery department and a research department, the first author (SRF) was responsible for data delivery from the databases (NPR and NRPHC), while the last author (IJB) had data management and research responsibilities. This setting is important as we covered the entire workflow from data extraction and data delivery to data management and use of data in research. By request at the data delivery department, research data is extracted from databases according to the study population and variable list specified by the researchers, ensuring that data is tailored to the project’s unique requirements (see Figure 1). The final data is extracted and stored in a suitable format, traditionally usually the text-based, unstructured CSV format. Since the data files included sensitive data and consisted of a large data set, data files need to be compressed and encrypted before delivery via a file sharing system. Drugs in Childhood and Reproductive Life holds approvals for updates until 2032. The purpose of the research project is to study drug use in general, and short-term and long- term safety related to drug use in the population, but with a special focus on safety during pregnancy and in children/adolescents. The study population is defined from a combination of the Norwegian Population Register and the Norwegian Medical Birth Registry (MBRN): all women and men registered in the Norwegian Population Register as born from 1944 onwards, supplemented with individuals only registered in the MBRN. The total population consists of more than 9 million individuals. Data for this population are extracted and delivered to the project from numerous data sources, including NRPHC, NPR, and NorPD. In winter of 2024, data from NPR, covering the period from 2008 to August 2024, and data from the NRPHC spanning from 2016 to August 2024, were delivered in the traditional row-based format CSV (Delivery 1). We began exploring alternatives to CSV for delivering large datasets and quickly selected the Parquet file format for further work. We identified numerous online resources, referencing only a few that we found most beneficial. 12 - 15 Importantly, we also engaged with scientists at other institutions experienced in handling large datasets to discuss whether Parquet could effectively address our CSV challenges (see acknowledgements). In spring 2025, a new data delivery was completed using the Parquet file format, which included updated data covering the entire year 2024 (Delivery 2). We will discuss the differences between CSV and Parquet and highlight the tools we have found most useful. This methodology improvement project was carried out in accordance with regulations of the NRPHC and NPR and did not require ethical approvement. The Drugs in Childhood and Reproductive Life Project was approved by the Committee for Medical and Health Research Ethics in Norway (REK South- East A; 2017/2546) and The Norwegian Data Inspectorate in Norway (17/02068/Norwegian Data Inspectorate). Results The first data delivery from NPR and NRPHC was provided in the traditional CSV format with data updated until August 2024, while the second delivery was in Parquet format with data updated until December 2024 ( Table 1 ). View this table: View inline View popup Download powerpoint Table. Comparison of data deliveries 1 and 2 in the Drugs in Childhood and Reproductive Life project Despite the increased information content in Delivery 2, with Parquet file format rather than CSV, the total file size for NPR and NRPHC data combined was reduced from 76.24 GB to 12.97 GB (-83%). Using looping techniques for splitting up data into smaller files became unnecessary when using Parquet, as Parquet data easily can be organized as Parquet file systems by taking advantage of the partitioning options. In our example, we partitioned data by year, but it is also possible to use multiple categorical variables for partitioning. Thus, while the overall structure for Delivery 1 was a set of CSV files, Delivery 2 consisted of a Parquet file system. Parquet file systems comprise multiple files that are interconnected through their intrinsic structure, offering an experience as if working with a single cohesive file. Partitioning not only simplifies data management but also accelerates data processing. For instance, a query targeting a single year will only access that specific part of the system, making the querying process highly efficient. In addition to partitioning, the columnar storage format in Parquet also allows for faster data retrieval and processing, meaning that in-memory tasks can be tailored to only using the data necessary, within the structure provided directly from the register owners. Since CSV files are text-based, they lack intrinsic metadata regarding data types. In contrast, Parquet files include additional metadata specifying the format of each column (string, date, integer, numeric), along with statistics for each column and further information about offset and size. 16 In our example, when extracting data from register databases for research purposes, the original data formats are preserved in the resulting Parquet file system. This allowed us to leverage the careful curation of the original databases. While the writing time to Parquet was comparable to CSV, Parquet’s intrinsic and advanced compression provided significant savings in time and resources on the register delivery side. Parquet files where password encrypted by using the standard 7-Zip program in the same way as the CSV files. 17 While data from NPR and NPRHC were delivered as Parquet files in Delivery 2, only CSV files were delivered from NorPD, representing all prescription fills in Norway from 2004 to 2024, split into yearly files. We converted these CSV files into a Parquet file system, partitioned by year, mirroring the structure used in the NPR and NPRHC deliveries. In this process, we used the DuckDB package in R and manually specified the data format for each relevant column. To minimize the programming workload, we included only a few key variables essential to the research project. This effort was rewarding, as it not only reduced the total file size from 146 GB to 6 GB but also provided access to powerful data handling tools not available for CSV files, optimizing our ability to efficiently query this large dataset. Since NorPD data were originally delivered as CSV files, we had the opportunity to compare processing time using a straightforward example: calculating the number of individuals in Norway with a prescription fill from 2004 to 2024 with two different methods . With the original CSV files, we read each file into memory using fread from data.table, extracted distinct personal identifiers, and then aggregated them to determine the total number of individuals over the entire period. In contrast, querying the Parquet file system using the arrow package required no reading into memory and allowing for a comprehensive query in a single operation. The difference in processing time was striking: 27 minutes using the CSV files and 28 seconds with the Parquet file system. Both methods yielded the same result. We provide the code for the processing time test in the Online Supplement. Discussion We found the advantages of using the Parquet file format evident from the outset. We quickly integrated Parquet into our daily routine. Furthermore, we implemented these changes across our respective departments within just a few months. Strengths and limitations of the present study The primary strength of this study is our extensive experience with data handling and management, both on the register holder side and on the research side. We successfully tested the transfer and utilization of research data from NPR and NRPHC as Parquet files and converted large CSV data to Parquet. The Parquet data structures from NPR and NRPHC served as excellent templates for converting the NorPD data, which was delivered as CSV. We have only worked with SQL and R, which represents a significant limitation of the current study. Implementation Parquet is an open-source data file format, making it accessible for everyone. It has become a de facto standard for storage and sharing large tabular data sets across a variety of systems and analytic tools. 18 However, we recognize that Parquet may present challenges in traditional register-based research. While Parquet file handling is supported in R and Python, using Parquet files in other statistical software like SPSS or Stata might not be straightforward, potentially presenting difficulties for researchers who prefer these programs. Beyond the potential challenges of adapting to new data handling tools, research projects, such as our case study, often require repeated data updates. Some researchers may argue that receiving data in a different format could disrupt established workflows and add complexity. However, in our view, the advantages of using Parquet greatly outweigh the initial extra effort required. In the online supplement, we provide examples of initial steps for handling Parquet file systems using the R package arrow. These methods are accessible and similar to those found in the widely used R package tidyverse. Conclusions We found that switching from the row-based, unstructured CSV format to the highly structured, columnar Parquet format significantly improved efficiency in terms of file size and processing time both at the data delivery side and at the research department site. Utilizing Parquet and associated R packages swiftly became an integral part of our routine workflow, and we recommend those working with large datasets in CSV to consider transitioning to Parquet, preferably in a program that will take the full advantage of the possibilities of this data structure. Another possibility is to learn the basic data management steps of Parquet data handling in R or Python, and then to continue the usual workflow. Data Availability Data are accessible to authorized researchers after ethical approval and application to https://helsedata.no/ . We also used simulated data that can be reproduced as described in the supplemental material. Declaration of conflicting interests The authors declared no potential conflicts of interest with respect to the research, authorship and/or publication of this article. Online Supplement 1 Tutorial: Arrow and Parquet Objective: To lower the entry barrier for those interested in working with Parquet files. We use a fake data set. Introduction Please check that packages arrow , data.table and dplyr have been installed. The first code block activates the necessary packages and generates a random data set with 1 million rows. Download figure Open in new tab We now designate file paths for a CSV file and a Parquet file and write the files to disk. Download figure Open in new tab After this step, you should have a new file in your working directory, “dataset.csv” and a new folder “dataset.parquet”. We can compare file sizes by inspecting the folder by using the file.size command. Download figure Open in new tab Note that the Parquet file size is just 22.7 percent of the CSV file size. We clean up the environment again: Download figure Open in new tab We now only use the Parquet file in the further work and start taking advantage of the arrow package. Download figure Open in new tab Now repeat the above command, piping in a select (notice logic of the code and the similarity to tidyverse ). Download figure Open in new tab Now we want to add a filter as well. Again, use pipe operations and inspect the result in the console and the environment: Download figure Open in new tab We want part of our data into the environment: The variables id , age and score for people above 80 years. Notice collect() as the last piped command: Download figure Open in new tab Now in the environment is a small fraction of the “original” data, selected three variables and filtered to a part of the population. Notice that the full data set never was read into the environment. Using data.table the same operation would have been: Download figure Open in new tab Or by in dplyr : Download figure Open in new tab It is time for cleaning up the environment again: Download figure Open in new tab Partitioning and compression Previously, we saved our fake data set to a Parquet file without partitioning. Now we aim to save data to a Parquet file system , partitioned by a categorical variable. Let’s begin by creating a simple fake data set, this time consisting of 10 million rows. Download figure Open in new tab We will save this data set as a Parquet file system, using Zstandard (ztsd) compression and partitioning by the variable for year of signup. Download figure Open in new tab We see that there is now a folder called “our_parquet_system” in the the directory. Let’s take a look at the content of the folder: Download figure Open in new tab We observe that there are eight subfolders, one for each sign-up year. We will now take advantage of the combined capabilities of arrow and Parquet in a few simple examples. First, check the overall content: Download figure Open in new tab We learn about the variables and the overall structure from the above command (without reading the file into memory). Now, we aim to conduct our first analyses, focusing on gathering the following information: The total number of distinct ID’s (although we already know the answer) The average weight by year The total number of distinct ID’s just for 2025 We gather this information as objects in the environment. Download figure Open in new tab We learnt that there were 10 000 000 people in the sample in total. In 2025, a total of 1 249 854 signed up. The weight was no surprise extremely stable over the years: Download figure Open in new tab View this table: View inline View popup Download powerpoint Conclusions We hope that this tutorial can help you get started with arrow and Parquet. Online Supplement 2 Source Code for our Processing Time Example Example: We checked processing time for calculating the total number of people in Norway with a prescription in Norway using 1) Arrow on a Parquet file system created from the delivered CSV files 2) data.table on CSV files delivered from the Norwegian Prescription Database. Run time was 28 minutes for the first approach and 27 seconds for the second approach. Please note that this supplement only gives the source code (cannot be run) Source Code for Speed Test Download figure Open in new tab Download figure Open in new tab Translating Data from CSV to Parquet using duckdb DuckDB is a bit more complicated to work with than arrow, but comes in handy when working with particularly large data sets. Please note that this part only gives example code and can’t be run. Download figure Open in new tab Download figure Open in new tab Download figure Open in new tab Acknowledgements We wish to thank IT-consultant Francesco Frassinelli and Dr. Sara Ghaderi at the Norwegian Tax Administration for discussions, insightful advice and encouragement in the initial phases of this work. References 1. ↵ Frank L. When an entire country is a cohort . Science 2000 ; 287 : 2398 – 2399 . OpenUrl FREE Full Text 2. ↵ Bakken IJ , Ariansen AMS , Knudsen GP , et al. The Norwegian Patient Registry and the Norwegian Registry for Primary Health Care: Research potential of two nationwide health-care registries . Scand J Public Health 2020 ; 48 : 49 – 55 . OpenUrl CrossRef PubMed 3. ↵ Furu K , Wettermark B , Andersen M , et al. The Nordic countries as a cohort for pharmacoepidemiological research . Basic Clin Pharmacol Toxicol ; 2010 ; 106 : 86 – 94 . OpenUrl CrossRef PubMed Web of Science 4. ↵ The Norwegian Institute of Public Health . Somatiske helsetjenester . 2024 . https://statistikk.fhi.no/npr/CJV3rajEfU8up3HsaoID-oJ5LQC0e-R_wxSjhpxYH4A . 5. ↵ The Norwegian Institute of Public Health . Statistikk og rapporter fra Kommunalt pasientog brukerregister (KPR) . 2024 . https://www.fhi.no/he/kpr/statistikk-og-rapporter/ . 6. ↵ The Norwegian Institute of Public Health . Legemiddelstatistikk per ATC-kode . 2025 . https://statistikk.fhi.no/lmr/bKQr_q4ImSChYfZzT1VwOeObMrXQ4K4j . 7. ↵ Laugesen K , Ludvigsson JF , Schmidt M , et al. Nordic health registry-based research: a review of health care systems and key registries . Clin Epidemiol 2021 ; 533 – 554 . 8. ↵ Ludvigsson JF , Håberg SE , Knudsen GP , et al. Ethical aspects of registry-based research in the Nordic countries . Clin Epidemiol 2015 ; 491 – 508 . 9. ↵ Hernán MA and Robins JM. Using big data to emulate a target trial when a randomized trial is not available . Am J Epidemiol 2016 ; 183 : 758 – 764 . OpenUrl CrossRef PubMed 10. ↵ Cohen JM , Cesta CE , Kjerpeseth L , et al. A common data model for harmonization in the Nordic Pregnancy Drug Safety Studies (NorPreSS) . Norsk Epidemiologi 2021 ; 29 . 11. ↵ The Norwegian Institute of Public Health . Epidemiological studies on benefits and harms of drug treatment – special focus on drug safety in pregnancy and childhood . 2025 . https://www.fhi.no/en/cristin-projects/ongoing/epidemiological-studies-on-benefits-and-harms-of-drug-treatment--special-fo/ . 12. ↵ Large Data in R: Tools and Techniques . 2021 . https://hbs-rcs.github.io/large_data_in_R/ . 13. Laderas T. Large Data Work: Intro to parquet files in R . 2024 . https://hutchdatascience.org/data_snacks/r_snacks/parquet.html . 14. Ilic N. Parquet file format – everything you need to know! 2024 . https://data-mozart.com/parquet-file-format-everything-you-need-to-know/ . 15. ↵ Apache . Parquet: Documentation . 2022 . https://parquet.apache.org/docs/ . 16. ↵ Apache . Parquet: Metadata . 2025 . https://parquet.apache.org/docs/file-format/metadata/ . 17. ↵ Apache . Parquet: Compression . 2024 . https://parquet.apache.org/docs/file-format/data-pages/compression/ . 18. ↵ Morusu V. Securing Parquet Files: Vulnerabilities, Mitigations, and Validation . 2025 . https://dzone.com/articles/securing-parquet-files-vulnerabilities-mitigations?fromrel=true#:~:text=Apache%20Parquet%20in%20Data%20Warehousing%20Parquet%20files,storage%20file%20format%20for%20large%2Dscale%20data%20processing. View the discussion thread. Back to top Previous Next Posted October 17, 2025. Download PDF Data/Code Email Thank you for your interest in spreading the word about medRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following Improve data management in register-based research: Transition from CSV to Parquet Message Subject (Your Name) has forwarded a page to you from medRxiv Message Body (Your Name) thought you would like to see this page from the medRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share Improve data management in register-based research: Transition from CSV to Parquet Simone Rahel Fenk , Kari Furu , Inger Johanne Bakken medRxiv 2025.10.15.25337992; doi: https://doi.org/10.1101/2025.10.15.25337992 Share This Article: Copy Citation Tools Improve data management in register-based research: Transition from CSV to Parquet Simone Rahel Fenk , Kari Furu , Inger Johanne Bakken medRxiv 2025.10.15.25337992; doi: https://doi.org/10.1101/2025.10.15.25337992 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Epidemiology Subject Areas All Articles Addiction Medicine (570) Allergy and Immunology (863) Anesthesia (301) Cardiovascular Medicine (4442) Dentistry and Oral Medicine (444) Dermatology (383) Emergency Medicine (609) Endocrinology (including Diabetes Mellitus and Metabolic Disease) (1511) Epidemiology (15232) Forensic Medicine (30) Gastroenterology (1126) Genetic and Genomic Medicine (6610) Geriatric Medicine (669) Health Economics (998) Health Informatics (4542) Health Policy (1370) Health Systems and Quality Improvement (1613) Hematology (543) HIV/AIDS (1266) Infectious Diseases (except HIV/AIDS) (15924) Intensive Care and Critical Care Medicine (1104) Medical Education (623) Medical Ethics (147) Nephrology (668) Neurology (6609) Nursing (346) Nutrition (999) Obstetrics and Gynecology (1146) Occupational and Environmental Health (957) Oncology (3338) Ophthalmology (974) Orthopedics (369) Otolaryngology (420) Pain Medicine (436) Palliative Medicine (130) Pathology (665) Pediatrics (1693) Pharmacology and Therapeutics (692) Primary Care Research (712) Psychiatry and Clinical Psychology (5450) Public and Global Health (9240) Radiology and Imaging (2203) Rehabilitation Medicine and Physical Therapy (1370) Respiratory Medicine (1197) Rheumatology (596) Sexual and Reproductive Health (714) Sports Medicine (530) Surgery (712) Toxicology (99) Transplantation (289) Urology (265) (function(){function c(){var b=a.contentDocument||a.contentWindow.document;if(b){var d=b.createElement('script');d.innerHTML="window.__CF$cv$params={r:'a020dc13288de748',t:'MTc3OTg0MTI4OA=='};var a=document.createElement('script');a.src='/cdn-cgi/challenge-platform/scripts/jsd/main.js';document.getElementsByTagName('head')[0].appendChild(a);";b.getElementsByTagName('head')[0].appendChild(d)}}if(document.body){var a=document.createElement('iframe');a.height=1;a.width=1;a.style.position='absolute';a.style.top=0;a.style.left=0;a.style.border='none';a.style.visibility='hidden';document.body.appendChild(a);if('loading'!==document.readyState)c();else if(window.addEventListener)document.addEventListener('DOMContentLoaded',c);else{var e=document.onreadystatechange||function(){};document.onreadystatechange=function(b){e(b);'loading'!==document.readyState&&(document.onreadystatechange=e,c())}}}})();

Text is read by the "Ask this paper" AI Q&A widget below. Extraction quality varies by source — PMC NXML preserves structure cleanly, OA-HTML may include some navigation residue, and OA-PDF can have broken hyphenation. The publisher copy (via DOI) is the canonical version.

My notes (saved in your browser only)

Ask this paper AI returns verbatim quotes from the full text · source: preprint-html

Answers must be backed by verbatim quotes from this paper's full text. Hallucinated quotes are dropped automatically; if no verbatim passage answers the question, we say so. How this works

Citation neighborhood (no data yet)

We don't have any in-corpus citations linked to this paper yet. This is a recent paper (2025) — citers typically take a year or two to land, and the OpenAlex reference graph may still be filling in.

Source provenance

europepmc
last seen: 2026-05-20T01:45:00.602351+00:00