First growth, then information: the path to genetic heredity in protocells

preprint OA: closed
📄 Open PDF Full text JSON View at publisher

Abstract

Most theoretical work on the origin of heredity has focused on how genetic information can be maintained without mutational degradation in the absence of error-proofing systems. A simple and parsimonious solution assumes the first gene sequences evolved inside dividing protocells, which enables selection for functional sets. But these models of information maintenance do not consider how protocells acquired their genetic information in the first place. Clues to this transition are suggested by patterns in the genetic code, which indicate a strong link to autotrophic metabolism, with early translation based on direct physical interactions between amino acids and short RNA polymers, grounded in their hydrophobicity. Here, we develop a mathematical model to investigate how random RNA polymers inside autotrophically growing protocells could evolve better coding sequences for discrete functions. The model tracks a population of protocells that evolve towards two essential functions: CO 2 fixation (which drives monomer synthesis and cell growth) and copying (which amplifies replication and translation of sequences inside protocells). The model shows that distinct coding sequences can emerge from random RNA sequences driving increased protocell division. The analysis reveals an important restriction: growth-supporting functions such as CO 2 fixation must be more easily attained than informational processes such as RNA copying and translation. This uncovers a fundamental constraint on the emergence of genetic heredity: growth precedes information at the origin of life.
Full text 66,318 characters · extracted from preprint-html · click to expand
First growth, then information: the path to genetic heredity in protocells | bioRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-M677548'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search New Results First growth, then information: the path to genetic heredity in protocells Raquel Nunes Palmeira , Marco Colnaghi , View ORCID Profile Andrew Pomiankowski , Nick Lane doi: https://doi.org/10.1101/2025.11.17.688785 Raquel Nunes Palmeira 1 Department of Genetics , Evolution and Environment, University College London , UK Find this author on Google Scholar Find this author on PubMed Search for this author on this site Marco Colnaghi 1 Department of Genetics , Evolution and Environment, University College London , UK 2 Vrije Universiteit Amsterdam , The Netherlands Find this author on Google Scholar Find this author on PubMed Search for this author on this site Andrew Pomiankowski 1 Department of Genetics , Evolution and Environment, University College London , UK Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Andrew Pomiankowski For correspondence: a.pomiankowski{at}ucl.ac.uk Nick Lane 1 Department of Genetics , Evolution and Environment, University College London , UK Find this author on Google Scholar Find this author on PubMed Search for this author on this site Abstract Full Text Info/History Metrics Preview PDF Abstract Most theoretical work on the origin of heredity has focused on how genetic information can be maintained without mutational degradation in the absence of error-proofing systems. A simple and parsimonious solution assumes the first gene sequences evolved inside dividing protocells, which enables selection for functional sets. But these models of information maintenance do not consider how protocells acquired their genetic information in the first place. Clues to this transition are suggested by patterns in the genetic code, which indicate a strong link to autotrophic metabolism, with early translation based on direct physical interactions between amino acids and short RNA polymers, grounded in their hydrophobicity. Here, we develop a mathematical model to investigate how random RNA polymers inside autotrophically growing protocells could evolve better coding sequences for discrete functions. The model tracks a population of protocells that evolve towards two essential functions: CO 2 fixation (which drives monomer synthesis and cell growth) and copying (which amplifies replication and translation of sequences inside protocells). The model shows that distinct coding sequences can emerge from random RNA sequences driving increased protocell division. The analysis reveals an important restriction: growth-supporting functions such as CO 2 fixation must be more easily attained than informational processes such as RNA copying and translation. This uncovers a fundamental constraint on the emergence of genetic heredity: growth precedes information at the origin of life. Introduction The emergence of genetic heredity was a critical step in the origin of life. Before the advent of genes, proto-living systems could only be a product of whatever chemistry was kinetically and thermodynamically favoured, even if part of a complex reaction network. Once genes existed, natural selection could begin to shape life. The full genetic apparatus must have been present in the last universal common ancestor (LUCA) as all life shares the same genetic code and translation apparatus ( 1 – 5 ). But the selective forces that enabled the evolution of such an intricate set of rules and machinery remain obscure. A key theoretical challenge in understanding this transition has been how information could be maintained before the evolution of the genetic apparatus. Maintaining information in a prebiotic scenario is particularly difficult because the copying-error rate had to be high in the absence of error-proofing enzymes ( 6 ). With large numbers of copying errors, maintaining a long genetic sequence is practically impossible, but error-proofing enzymes are today invariably encoded by long genetic sequences. This is Eigen’s paradox: no error-proofing enzymes without long sequences, no long sequences without error-proofing enzymes ( 7 , 8 ). While Eigen’s paradox could be escaped if information is contained in groups of short sequences rather than one long one, the persistence of such groups is restricted by competition for resources (such as monomers, or catalysts for copying) leading to competitive exclusion. The question of how such sequences arose in the first place is also unresolved. Selection acting on RNA strands in solution favours those that are small and fast-replicating, “Spiegelman’s little monsters”, rather than sequences encoding other functions such as metabolism ( 9 ). Various models of the origin of heredity have addressed the problem of information maintenance ( 10 – 13 ). Among these, the simplest and most parsimonious proposal is for early sequences to evolve within dividing protocells, a solution advanced by the stochastic corrector model ( 14 ). This model solves the problem of information maintenance through selection at the level of the protocell. Fitness is highest if the protocell maintains a specified ratio of replicator types, but declines to zero if one of the essential replicator populations falls extinct. The proliferation of cells with the optimal balance of replicator sequences allows multiple sequences to co-exist, increasing the information content that can be maintained in a protocell. In contrast, selfish fast-replicating sequences are restricted by selection at the level of the protocell, as selfish replicators copy themselves at the expense of other sequences needed for protocell division. Selection at higher levels, such as the whole cell, to overcome selfish replicators is seen as a fundamental underpinning of Major Transitions in Evolution ( 8 ). But some questions were not addressed by the stochastic corrector and related models of early heredity (Michod, 2015, Boerlijst and Hogeweg, 1991, Bresch et al., 1980, Zachar et al., 2011). These models provide a basis for the maintenance of information in protocells, but leave open the question of how protocells acquired functional sequences in the first place, or what functions these first sequences encoded. The model we present in this paper seeks to span this gap by considering how protocells first acquired sequences that enhanced the two fundamental functions enabling evolution: differential fitness and heredity ( 15 ). In protocells, differential fitness could plausibly emerge from sequences that enhance protocell growth, for example by encoding peptides that catalyse CO₂ fixation. Heredity, in turn, relies on copying linked to translation, which allows catalytic functions to be transmitted over generations. We assume that protometabolic flux is catalysed by metal ions and ‘naked’ cofactors (see Discussion), which means that better CO 2 fixation would drive growth by generating more monomers, including fatty acids (for protocell membranes), amino acids (for peptides) and nucleotides (for RNA). The assumption of functional peptides requires the inheritance of RNA sequences that are both copied and ‘translated’ into catalytic peptides, so the fastest growing protocells propagate their more functional RNA. At the origin of heredity, RNA sequences must have arisen by chance and given rise to a diverse pool of translational products. We assume the most rudimentary form of translation followed simple rules based on hydrophobicity, grounded in patterns in the genetic code which suggest that interactions between amino acids and cognate nucleobases were key in early evolution ( 16 , 17 ). These behaviours are determined by simple physical chemistry—folding, partitioning, and binding to catalytic cofactors emerge spontaneously from polymer composition ( 18 – 21 ), which is inherited from RNA templates. The likelihood that protocells gained peptides conferring fitness would have depended on whether they were encoded by short or long sequences and by simple or more specific motifs. This raises the question of which of the two critical functions – growth or information – was more likely to arise first and permit the evolutionary expansion of protocells. To tackle this question, we develop a model of protocells with rudimentary heredity. We assume that RNA sequences can be copied and loosely translated into peptide sequences with enzyme functions that enhance either growth or information propagation. Specifically, the model tracks protocells containing random sequences that can evolve towards two explicit functions: catalysis of CO₂ fixation, which increases monomer production and thus cell growth, or catalysis of templated polymerisation of RNA and peptides (copying and translation). The ability of a peptide to perform one of these functions depends on its relative hydrophobicity, which determines its likelihood of folding and partitioning to either the membrane or cytosol ( 18 – 20 ). We assume that hydrophobic peptides tend to associate with the membrane, where they can catalyse CO₂ fixation ( 21 , 22 ), while hydrophilic peptides remain in the cytosol, where they can facilitate templated polymerisation through binding to metal ions such as Mg 2+ , as in modern enzymes ( 23 – 25 ). We then vary the optimal peptide compositions required for each function to ask if the RNA encoding them can evolve towards optimal sequences. We emphasise this means evolution from random RNA sequences, with zero intrinsic information, to biologically meaningful sequences encoding growth and copying peptides – the emergence of genetic information. The model shows that these simple rules enable the evolution of functional sequences from random distributions of nucleotides and peptides, and elucidates the parameter ranges that permit protocell growth. We establish a necessary condition for this early sequence evolution to occur in autotrophic protocells: sequences encoding catalysts of CO 2 fixation must evolve first, as they drive the addition of monomers and, in turn, increase the rate of random polymerisation and the ability of protocells to explore the sequence space, allowing the later evolution of sequences that encode information supporting functions. We suggest this observation is generalisable: as a rule, growth must come first. Model overview Protocell contents The model follows in discrete time the evolution of a population of protocells containing monomers and polymers of nucleotides and amino acids. A number of simplifying assumptions are made to allow general conclusions to be drawn. Nucleotide monomers are split into two classes, purines or pyrimidines. Purines, with their double ring structure, are somewhat more hydrophobic than pyrimidines, which have a single ring ( 26 ). This difference is captured by considering purines to be hydrophobic and pyrimidines to be hydrophilic. Likewise, amino acids are classed as hydrophobic (such as valine or alanine) or hydrophilic (such as aspartate and glutamate). The linear sequence of polymers made of nucleotides (RNA) and amino acids (peptides) are not explicitly considered. Instead, we keep track of the length ( l ) and relative hydrophobicity ( h ) of each polymer. The relative hydrophobicity of an RNA polymer is the difference between the number of hydrophobic purine monomers ( n R ) and hydrophilic pyrimidine monomers ( n Y ) divided by the length of the polymer, and for peptides as the difference between the number of hydrophobic amino acid monomers ( m ɸ ) and hydrophilic amino acid monomers ( m ζ ) divided by the length of the polymer, Hence, an RNA polymer composed entirely of purines has ℎ n = +1 (conversely pyrimidines ℎ n = ―1), and a peptide composed entirely of hydrophobic amino acids has ℎ p = +1 (conversely ℎ p = ―1). Protocell dynamics At each time step, the populations of monomers and polymers within a protocell are modified by five processes, in the following order ( Fig 1 ). ( 1 ) Nucleotide and amino acid monomers are added through CO 2 fixation. Flux from the first products of fixed CO 2 (carboxylic acids) through a non-genetically encoded protometabolism is assumed to form amino acids and nucleotides. In the model, the numbers of nucleotides ( n ) and amino acids ( m ) added are sampled from a binomial distribution with probabilities p n and p m with n m a x and p m a x trials. It is assumed that hydrophilic and hydrophobic nucleotides and amino acids are made in equal numbers. ( 2 ) Nucleotide monomers undergo polymerisation with probability p p , to form either random RNA dimers or to increase the length of existing RNA polymers (this step does not apply to amino acid monomers). ( 3 ) Copying and ( 4 ) translation of RNA polymers (≥ 2 monomers) are independent processes that happen separately. Copying occurs via base-pairing, with the resulting RNA copy having the same length ( l ) but opposite relative hydrophobicity (― h ) to the template. This reflects the reciprocal strand having the opposite number of purines and pyrimidines to the template. Translation is assumed to be based on hydrophobicity, with the resulting peptide having the same length ( l ) and relative hydrophobicity as the RNA template (i.e. h p = h n ). For both processes, all RNAs are sampled as potential templates with a probability p c for copying and p t for translation. The sampled templates are listed randomly and copied (or translated) in order until all nucleotide and amino acid monomers, respectively, are exhausted. This assumption means the rates of copying and translation are limited by monomer supply. Incomplete RNA copies and peptide transcripts are disregarded. ( 5 ) Finally, decay occurs in which all polymers (both RNA and peptides) have a probability p d of losing a monomer. This probability is weighted by the polymer length, as longer polymers have more bonds that could be broken. This assumption means that the monomer pool generated by CO 2 fixation is supplemented by monomers derived from polymer decay. We appreciate that this form of polymer decay is unrealistic, but from a modelling point of view it captures two important features quite simply: it limits the length of polymers and therefore the difficulty of replicating them, and it provides an additional supply of monomers for polymerization that does not depend directly on CO 2 fixation. Download figure Open in new tab Fig 1. Processes occurring in the protocell at each time step. Continuous arrows show polymers being produced, segmented arrows show peptide catalysis of CO 2 fixation and templated polymerisation, and dotted arrows indicate RNA acting as templates for copying and translation. Note that sequences shown here are illustrative, and the model only keeps track of hydrophobicity and length of sequences, and not sequence order. Catalysis Peptides are assumed to perform two functions. They catalyse templated polymerisation, enhancing copying of RNA and translation of RNA into peptides (red dashed arrows, Fig 1 ).They also catalyse CO 2 fixation, resulting in the addition of more nucleotide, amino acids and fatty acids used for protocell growth (blue dashed arrow, Fig 1 ). An alternative model where monomer production is largely unlinked to protocell growth is also considered (see section on heterotrophic growth below). Longer polymers are assumed to have greater catalytic power, which plateaus at longer lengths (Fig S2). The parameter k l represents the maximum catalytic power and ℽ represents the length at which catalytic power is at half maximum. Higher values of ℽ indicate that the optimal catalytic power is achieved by longer peptides, so we use this parameter as a proxy for the optimal length of a sequence. Hydrophobic peptides are assumed to preferentially partition to the membrane ( 20 ), where they catalyse CO 2 fixation, similar to the Energy-converting hydrogenase in methanogenic Archaea, ( 27 , 28 ). In contrast, hydrophilic peptides tend to remain in the cytosol, where they bind metal ions and catalyse the templated polymerisation of RNA and peptides. Given the deep differences between RNA and peptide synthesis, it might seem unjustified to group them as a single function performed by the same class of catalysts, but the RNA polymerase and ribosome both require Mg 2+ for their catalytic function ( 29 – 31 ), and both processes condense activated monomers by eliminating pyrophosphate, so the physical chemistry is in part equivalent. The relationship between peptide relative hydrophobicity and catalysis is modelled as a Gaussian curve with an optimum catalytic power at a specific relative hydrophobicity (Fig S3). The relative hydrophobicity that result in optimal catalysis of CO 2 fixation and templated polymerisation is given by the constants β fix and β pol (Fig S3). The β parameters were chosen so that the peaks of these curves do not overlap – so, some peptides might have low catalytic activity for either function, but no peptides have high catalytic activity for both functions. Growth, division and selection Protocells are assumed to grow at a rate proportional to that of CO 2 fixation. This follows from the assumption that the rate of fatty-acid synthesis (constituting new protocell membrane) is necessarily proportional to that of CO 2 fixation in autotrophic protocells. The same logic applies for nucleotides and amino acids, which are produced at a rate proportional to CO 2 fixation. When a protocell reaches an arbitrary size threshold, set by the number of fatty acids in the protocell, we assume it divides, and one other protocell is deleted at random from the population, in a similar process to a Moran model ( 32 ). The RNA and peptide polymers and monomers are divided between the two daughter cells with proportions x and 1 ― x , where x is sampled from a normal distribution truncated between 0 and 1, with mean 0.5 and standard deviation 0.2. In each simulation, 100 protocells are allowed to evolve for 10,000 time steps, where a time step is one ‘turn’ of the model (i.e. the 5 sequential processes in Fig 1 ). At the end of each time step, a protocell can divide if it has reached the threshold number of fatty acids for replication ( S ). We investigate how the rate of protocell division changes for different parameters and features of the model. In the model, selection arises simply from the growth rate, favouring faster rates of protocell growth. A full description of the model is given in the Supplementary Information. Parameter values A simplified model where optimal catalytic sequences are symmetrical was used to investigate the effect of each parameter on the dynamics. In the simplified model, the relationship between the length and the relative hydrophobicity of amino acid polymers on catalytic power is the same for CO₂ fixation and templated polymerisation. This leads to the unrealistic feature that a “Watson” nucleotide polymer that favours CO₂ fixation when copied automatically generates a “Crick” nucleotide polymer that favours templated polymerisation to the same extent. This model is only used to set parameter values that lead to protocell growth, used as the base values in Table 1 . We report variation in these parameters in the Supporting Information. For instance, increasing the baseline probabilities of copying and translation ( p c , p t ) and decreasing the probability of decay ( p d ) leads to higher catalytic rates and protocell growth (Figs S5-8). View this table: View inline View popup Download powerpoint Table 1. Parameters used in simulations (unless otherwise stated) Heterotrophic growth An alternative model where protocells are heterotrophic is also considered. In this model protocell growth and monomer addition are unlinked. Here, hydrophobic catalysts increase the number of fatty acids used for protocell growth but do not affect the intake of nucleotides or amino acids. Both nucleotides and amino acids are taken up from the environment at constant rates p n and p aa , respectively. Note that the CO 2 fixation catalyst was replaced with a growth catalyst rather than a catalyst enhancing intake of monomers because a function that directly links to protocell division is still necessary. Otherwise, the lack of a growth catalyst would render selection at the level of the protocell impossible. Results Evolution requires CO 2 fixation to be easier than copying Reliable evolution of protocells with high rates of division is possible when the optimal sequences for CO 2 fixation are easier to generate at random than those for templated polymerisation. Specifically, when CO 2 fixation catalysts are shorter (γ fix < γ pol ) and weakly hydrophobic (|β fix | < |β pol |) they arise more easily by chance and provide an initial selective advantage. In this case, the rate of protocell division quickly and consistently evolves to a high level (blue line, Fig 2A ). The increased rate of CO 2 fixation results in more monomers being generated per time step, increasing the probability of random polymerisation. This allows protocells to explore sequence space and generate the less probable RNA sequences that encode catalysts for templated polymerisation. Selection then gives rise to protocells with RNA polymers that encode both catalytic optima, one for CO 2 fixation and the other for templated polymerisation ( Fig 2B ). This leads to a positive feedback loop, with selection favouring high rates of protocell growth ( Fig 2A ). Download figure Open in new tab Fig 2. Evolution of protocells with different catalytic optima for CO₂ fixation and templated polymerisation. Panel (A) shows the moving sum of protocell divisions over 50 time steps, and panels (B–C) show the average RNA compositions of protocells at time step 8000. Catalytic rates ( k l and k ℎ ) depend on the optimal catalytic length (γ) and hydrophobicity (β) for CO₂ fixation (γ fix , β fix ) and for templated polymerisation (γ pol , β pol ). We compare two contrasting scenarios in which either CO₂ fixation or templated polymerisation catalysts are easier to achieve at random, due to differences in their optimal γ and β values. In panel (B) (blue line in panel a), CO₂ fixation catalysts are easier to evolve, with γ fix = 6, γ pol = 8, β fix = 0.2, and β pol = –0.8. In panel (C) (orange line in panel a), templated polymerisation catalysts are easier to evolve, with γ fix = 8, γ pol = 6, β fix = 0.8, and β pol = –0.2. Each square in the heatmaps shows the log count of RNA molecules with a given number of purines and pyrimidines. All other parameters are listed in Table 1 . In contrast, when RNA strands coding for catalysis of templated polymerisation are shorter (γ fix > γ pol ) and weakly hydrophilic (|β fix | > |β pol |), protocells with high rates of division never evolve (orange line, Fig 2A ). In this case, catalysts for templated polymerisation arise at random more easily and increase the rate of copying and translation. But that makes it less likely that longer sequences with more extreme hydrophobicity needed to catalyse CO 2 fixation arise by chance. The distribution of RNA strands under this condition is little better than random ( Fig 2C ). Without the production of RNA strands encoding catalysts of CO 2 fixation the monomer supply remains limited, restricting protocell growth. We also explored simulations in which one of the parameters (RNA length or relative hydrophobicity) was more extreme for CO 2 fixation while the other was more extreme for templated polymerisation. Under these intermediate conditions, neither one sequence nor the other is favoured to arise at random more strongly than the other. Evolution towards higher protocell division rates can occur but at a low frequency (1-3 % of simulations, Fig S9). This occurs in the unlikely case that RNA strands coding for CO 2 fixation do form and initiate a selective advantage through greater input of monomers. Only then is it beneficial for the protocell lineage to evolve better templating (i.e. copying and translation) as well. Decoupling growth and monomer supply relaxes constraints on growth The results above depend on the assumption that protocells are autotrophic, making organics by reducing CO 2 . In the model, the rate of protocell growth is proportional to the rate of monomer addition. This favours sequences that raise protocell fitness by increasing monomer input, with the downstream effect that sequence space is better explored, enhancing catalytic functions. If the link between protocell growth and monomer availability is broken, these feedbacks no longer apply as strongly. If monomers are delivered at high rates from an exogenous ‘soup’ (implying a heterotrophic origin), then sequence space will inevitably be explored, albeit ‘easy growth’ is still slightly favoured over ‘easy templated polymerisation’ ( Fig. 3A ). Protocells evolve towards higher division rates irrespective of whether optimal catalytic sequences are easier to generate at random for CO 2 fixation (γ fix < γ pol and |β fix | γ pol and |β fix | > |β pol |; Fig 3C ). The high rate of monomer input (specifically amino acids and nucleotides) means that even if catalysis of templated polymerisation evolves first, this does not deplete monomers sufficiently to hinder the production by random nucleotide polymerisation of RNA molecules that code for peptides catalysing growth. In other words, growth is still required for the evolution of functions beyond RNA replication, but is not needed for the cell to explore sequence space through random polymerisation. Download figure Open in new tab Fig 3. Evolution of protocells with different catalytic optima for growth and templated polymerisation, when growth catalysis is unlinked to monomer supply and p n = p aa = 0 . 1 . Panel (A) shows the moving sum of protocell divisions over 50 time steps, and panels (B–C) show the average RNA compositions of protocells at time step 8000. Catalytic rates ( k l and k ℎ ) depend on the optimal catalytic length (γ) and hydrophobicity (β) for growth (γ g , β g ) and for templated polymerisation (γ pol , β pol ). We compare two contrasting scenarios in which either growth or templated polymerisation catalysts are easier to achieve at random, due to differences in their optimal γ and β values. In panel (B) (blue line in panel a), growth catalysts are easier to evolve, with γ g = 6, γ pol = 8, β g = 0.2, and β pol = –0.8. In panel (C) (orange line in panel a), templated polymerisation catalysts are easier to evolve, with γ g = 8, γ pol = 6, β g = 0.8, and β pol = –0.2. Each square in the heatmaps shows the log count of RNA molecules with a given number of purines and pyrimidines. All other parameters are listed in Table 1 . Critically, this latter observation only holds true if monomer supply is high enough for protocells to freely explore sequence space, such that RNA copying and random polymerisation do not compete for monomers ( Fig 4 , p n = p aa = 0.01). If monomer supply is limiting, then protocells never sharply increase their division rates ( Fig 4 , p n = p aa Download figure Open in new tab Fig 4. Decoupling growth and monomer supply. Evolution of protocells when growth catalysis is unlinked to monomer supply with variable rates of nucleotide and amino acid monomer addition ( p n and p aa ). Panels (A) and (B) show the moving sum of protocell divisions over 50 time steps for decreasing rates of monomer addition ( p n = p aa = 0.01, 0.001, 0.0001 and 0.00001), higher rates of monomer addition are shown in darker colours and lower rates are shown in lighter colours. Panel (A) shows the scenario when growth catalysts are easier to evolve (γ g = 6, γ pol = 8, β g = 0.2, and β pol = –0.8). Panel (B) shows the scenario when growth catalysts are easier to evolve (γ g = 6, γ pol = 8, β g = 0.2, and β pol = –0.8). Note that p n = p aa = 0.001 corresponds to the monomer availability driving robust growth in the autotrophic model, but produces very limited growth here. All other parameters are listed in Table 1 . = 0.001, 0.0001, and 0.0001). Specifically, at intermediate rates of monomer supply, roughly equal to the baseline monomer supply for the autotrophic model growth remains low in all simulations ( Fig 4 , p n = p aa = 0.001). While the model was not intended to directly compare autotrophic versus heterotrophic origins, it is worth noting that heterotrophic growth rates are substantially lower than autotrophic growth rates. That happens because selection for protocell growth in the autotrophic model also leads to increased monomer synthesis, and more monomers generate more catalysts, forming a positive feedback loop. This loop is not possible under the heterotrophic model. Random polymerisation does not rescue growth when CO₂ fixation is disfavoured Considering again an autotrophic model (where growth and monomer supply are fully linked), a higher probability of random polymerisation potentially could alleviate the requirement for CO₂ fixation to evolve more easily than templated polymerisation. To test this idea, the random polymerisation of nucleotides ( p p ) was raised in simulations where catalysis of CO₂ fixation was less likely to arise by chance than templated polymerisation (γ fix > γ pol and |β fix | > |β pol |). The rate of protocell division does in fact increase slightly at higher probability of random polymerisation ( p p = 0.01 - 0.08; Fig 5A ), as catalysis of CO₂ fixation arises more often by chance. However, further rises in the probability of random polymerisation do not increase the rate of protocell division, and overall there is little fitness advantage ( p = 0.12; Fig 5A ). This is because higher values of p p also make random polymerisation more dominant, which dilutes any high-fitness RNA sequences with random noise ( Fig 5B-E ). As a result, the system fails to generate the positive feedback loop described earlier, in which catalysis of CO₂ fixation enhances monomer supply, which in turn improves copying, again amplifying CO₂ fixation. Without this feedback loop, RNA distributions never centre around the catalytic optima ( Fig 5B–E ) and protocell divisions always remain low ( Fig 5A ) – unlike cases where CO₂ fixation is favoured and distributions become centred on functional sequences ( Fig 2B ). Download figure Open in new tab Fig 5. The effect of varying the probability of random polymerisation ( p p ). This is investigated when templated polymerisation catalysts are easier to achieve at random (γ fix = 8, γ pol = 6, β fix = 0.8 , and β pol = –0.2.). (a) Evolutionary change in the rate of protocell divisions (per 50 time steps) for p p = 0.01 (blue), p p = 0.04 (orange), p p = 0.08 (yellow) and p p = 0.12 (purple). The mean distribution of nucleotides across protocells at t = 8,000 (red dotted line in panel (A)) is shown for (B) p p = 0.01, (C) p p = 0.04, (D) p p = 0.08 and (e) p p = 0.12. Each square in the heatmap shows the log count of RNA molecules composed of specific numbers of purines and pyrimidines. All other parameter values are in Table 1 . In the complementary case, where catalysis of CO₂ fixation is favoured (γ fix < γ pol and |β fix | < |β pol |), protocell division rates fall modestly but consistently with increasing probabilities of random polymerisation (Fig S10). The same dynamics apply here: random polymerisation reduces the relative advantage of templated copying and disrupts the maintenance of catalytic RNAs. Although protocell growth is still possible, the efficiency of selection is reduced by noise. Discussion and conclusions The backdrop to the question of the origin of genetic heredity and evolution can be framed in terms of the basic requirements for evolution by natural selection ( 15 ). Given rudimentary heredity and natural variation produced by stochasticity, what should be favoured first: differential fitness (protocell growth) or reliable heritability (templated polymerisation)? This might appear to be a classic ‘chicken-and-egg’ dilemma, but the mechanistic details of protocell function provide insight. In the model, the two functions are encoded by distinct sequences. An important outcome is that sequences encoding CO₂ fixation for growth must be synthesised more readily, and evolve before those encoding templated polymerisation. In other words, differential fitness must arise first, and reliable heritability second. The reason lies in the fact that, in autotrophic protocells, ‘protocell fitness’ is intrinsically linked to the production of new sequences – CO₂ fixation produces membrane fatty acids but also nucleotides and amino acids. If the best catalysts for CO₂ fixation are short and have modest specificity of hydrophobicity, they will be more easily generated at random and promote higher rates of protocell division ( Fig 2A ). The resulting increase in CO₂ fixation promotes monomer production, which in turn drives higher rates of polymerisation, enabling protocells to explore RNA sequence space, and eventually producing longer, more specific hydrophilic catalysts that support templated polymerisation ( Fig 2B ). The resulting positive feedback loop leads to improved copying and translation alongside CO₂ fixation. By contrast, if templated polymerisation catalysts arise first – because they are shorter or more weakly hydrophobic – then they consume nucleotide monomers, making it harder for protocells to synthesise the longer, more hydrophobic catalysts needed for CO₂ fixation ( Fig 2C ). As a result, protocell growth never takes off. We can draw a general conclusion from this finding: differential fitness must precede reliable heritability, and growth precede copying. This conclusion rests on the link between monomer supply and protocell growth. If the CO₂ fixation catalyst is replaced by a ‘growth catalyst’ – one that only promotes fatty-acid synthesis or scavenging – and the nucleotide and amino-acid monomers are added at a steady, constant rate, the results change. This version of the model breaks the assumption of autotrophy and mimics a heterotrophic protocell in which monomer supply is externally fixed and uncoupled from RNA encoded catalysis. Under these conditions, and with a high rate of external monomer supply, either class of catalysts – those for growth or for templated polymerisation – can evolve first, and protocells still achieve high division rates ( Fig 3A ). Because monomers are no longer limiting, the evolutionary order of functions is less constrained. But in the more likely case that external monomer supply is steady but limiting – equivalent to baseline monomer availability in the autotrophic model – protocells evolve more reliably to slightly higher division rates when growth catalysts are easier to evolve at random ( Fig 4 , p n = p aa = 0.01). If the monomer supply from the environment is even lower, protocell growth fails to take off at all ( Fig 4 , p n = p aa = 0.001). While the availability of monomers in any ‘primordial soup’ is hard to predict, uniformly high concentrations seem implausible. If all types of monomers were not uniformly available (which would likely be the case for hydrophobic amino acids and nucleotides) then rates of growth would collapse to those supported by the availability of scarcest monomer. Worse, the heterotrophic model still requires either de novo fatty-acid synthesis from CO 2 fixation (no easier than the other pathways assumed in the autotrophic model) or some form of differential scavenging of fatty acids from the environment by the growth catalyst. Given the physical tendency of fatty acids to incorporate directly into fatty-acid bilayers, it is hard to see how such a catalyst could benefit some cells over others, in which case cell-level selection would not be possible. In the modelling, the heterotrophic model serves as a valuable proof of concept. A reliable form of heredity can evolve independent of which peptide function arises first but only when monomer supply is abundant and does not limit sequence formation. This condition seems implausible. Our choice to focus on autotrophic protocells is justified by phylogenetic and experimental data. Phylogenetic reconstructions show that LUCA was likely chemolithoautotrophic converting CO 2 and H 2 into organics ( 1 , 5 , 33 ), and experimental work has shown entire sections of autotrophic metabolism are possible under plausible prebiotic conditions ( 22 , 34 – 40 ). Patterns in the genetic code provide further support: the first base of the codon correlates with the biosynthetic distance of the amino acid from CO₂ fixation, with amino acids closer to the metabolic core encoded by purines and more distal ones by pyrimidines ( 17 ). These observations strongly suggest that the genetic code arose in the context of an expanding autotrophic metabolism rooted in CO₂ fixation. A seemingly alternative way autotrophic protocells could have explored large regions of sequence space would be high rates of random polymerisation. This increases the likelihood of generating functional sequences by chance, potentially bypassing the need for CO₂ fixation catalysts to arise first. To test this, the probability of random polymerisation ( p p ) was varied. However, the effect of increasing p p was limited or detrimental. Increase in random polymerisation allows easier exploration of parameter space but it also increases competition for monomers needed for copying. This reduces the amplification of beneficial sequences thereby stalling growth ( Fig 5A , Fig S10). Protocells are only able to accumulate functional sequences when copying is capable of outcompeting noise (i.e. random polymerisation). This reveals an important asymmetry. Higher values of random polymerisation increase the rate at which random sequences form, but does not feed through to an increase the total monomer pool. In contrast, increasing monomer input through raised CO₂ fixation expands the pool of available monomers and allows copying, once established, to dominate over random processes. In this way, growth enables information not just by increasing exploration, but by increasing the supply of monomers that copying can exploit, allowing beneficial sequences to be selectively amplified. Although increased random polymerisation did not alter the overall pattern of results, it did decrease the ability of protocells to retain sequence distributions centred around catalytic optima ( Fig 5C-E ). Some degree of random polymerisation is necessary to explore sequence space, but excessive rates reduce the efficiency of selection and hinder the maintenance of functional sequences. This is a standard result of population genetics, noise reduces the efficacy of selection ( 41 ). It also seems plausible that templated polymerisation occurs at a higher rate than random polymerisation. De novo polymerisation of nucleotides in water is difficult. The only successful attempts have used wet-dry cycles ( 42 ), eutectic freezing ( 43 ), thermal gradients ( 44 ), and basalt rock glasses ( 45 ), but the relevance of these conditions to protocells is ambiguous as both RNA and peptide polymerisation involves monomer activation and elimination of pyrophosphate ( 46 , 47 ). There have been fewer attempts to elucidate the conditions needed for templated polymerisation of nucleotides. This has also proved difficult, and the only successful attempts were achieved using ‘pre-condensed’ monomers (nucleoside phosphorimidazoles), which are not consonant with life ( 48 ). While both reactions are clearly difficult, templated polymerisation should enable monomers to bind to the template through hydrogen bonding and stacking interactions ( 49 , 50 ), which partially immobilizes them, increasing the likelihood of condensation to form a polymer. It is then credible that modest catalytic enhancement – whether by cofactors, ribozymes, or short peptides, simple precursors of the Mg 2+ -dependent RNA polymerase, as modelled here – could have tipped the balance in favour of copying over random polymerisation. There are good reasons to think that CO 2 fixation can indeed evolve more easily than templated polymerisation, as there are many ways to enhance CO 2 fixation. The phrase ‘CO 2 fixation’ implicitly includes all intermediary metabolism between the initial fixing of CO 2 through to the synthesis of fatty acids, amino acids and nucleotides. Catalysts of CO 2 fixation therefore include any catalysts that drive flux through intermediary metabolism, and consequently cell growth. A caveat to this is that flux through the metabolic network must be balanced. Our earlier work shows that strong catalysis of a single pathway (e.g. the synthesis of sugars or amino acids) diverts flux down one pathway at the expense of others, unbalancing metabolism and slowing down protocell growth and division ( 51 ). As such, catalysts of CO 2 fixation include promiscuous ‘naked’ cofactors (which function in the absence of an enzyme, just slower) that catalyse various metabolic pathways simultaneously, such as the transfer of hydride ions by NADH or equivalents ( 51 – 53 ). CO 2 fixation can be facilitated by NADH in acetogenic bacteria ( 54 ), by [4Fe-4S] clusters in Ech or ferredoxin ( 27 , 55 , 56 ), by Mo or Ni in CODH ( 57 , 58 ), and by nucleotide cofactors such as pterins and folates in methanogens and acetogens ( 33 , 59 , 60 ). Conversely, RNA polymerisation involves only one specific repeated reaction, and all polymerases have a conserved functional core ( 61 ). To uncover the conditions favouring this evolutionary transition to the emergence of genetic information, we drew on patterns in the genetic code, which predict that translation emerged through direct physical interactions between amino acids and the RNA bases that code for them. Associations guiding codon assignment have been documented since the 1960s (Crick, 1968, Woese, 1965, Wong, 1975, Lacey et al., 1984), and amount to the so-called ‘code within the codons’ (Taylor and Coates, 1989, Copley et al., 2005). The clearest pattern links the hydrophobicity of the RNA anticodon middle base with that of the cognate amino acid (Harrison et al., 2022). This connection suggests that early translation was based on direct interactions between anticodon bases and amino acids with similar hydrophobicity, or more subtly, partition energy (Caldararo and Di Giulio, 2022). The existence of direct physical interactions is supported by miscellaneous experimental and computational work (Lacey et al., 1984, Shimizu, 1987, Shimizu, 1995, Hobish et al., 1995, Yarus et al., 2009), albeit this has remained controversial ( 62 ). Recently, systematic molecular dynamics simulations showed that half of all proteinogenic amino acids interact most strongly with the cognate middle base of the anticodon, and this general trend was confirmed by NMR (Halpern et al., 2023). If these interactions were at the heart of coding, then early genetic information would have been simple and coarse grained, producing what Woese termed ‘statistical proteins’ (Woese, 1965). Coding would not have been based on the sequence of specific nucleotides, but on their general hydrophobicity (or as noted, related interactions). Because purines are more hydrophobic than pyrimidines, a succession of purines in RNA would template a relatively hydrophobic peptide through direct interactions between amino acids and cognate bases. This assumption underpins our model. The assumption that early translation was based on hydrophobicity provides a bridge between the structure of the genetic code and the emergence of functional peptides in an autotrophic protocell context. In principle, the same logic could apply to a model without peptides, in which RNA sequences alone acted as catalysts. If RNA sequences folded to catalyse both CO₂ fixation and their own replication, selection would still favour the same ordering of functions. But this scenario comes with an unhelpful trade-off: ribozymes that fold into complex catalytic structures are likely to be harder to copy, as their complex secondary structure needs to be unpicked, creating a tension between their catalytic performance and replicability. The current model does not need to account for that trade-off, as peptides are responsible for all functions. In any case, even if these functions are carried out by ribozymes rather than peptides, the overall result would still hold: CO₂ fixation needs to arise before reliable heredity, whether catalysed by peptides or RNA, so long as it fuels monomer production and supports sequence exploration and selective amplification. Of course, the assumption that CO 2 fixation (or other core metabolic functions) is catalysed by ribozymes rather than enzymes is ungrounded in empirical observation (there are no known examples in life) and even less helpfully, defers the question of how proteins came to replace ribozymes, which is solved simply here through direct physical interactions. We have therefore focused on loosely templated translation and the evolution of Woeseian ‘statistical’ proteins. For simplicity, neither version of the model considers loss or gain of polymers from the environment. Full polymers were lost only when a cell divides and another in the population was deleted at random. This assumption is consistent with the fact that, in modern cells, specific transporters are needed to import nucleosides and amino acids ( 63 ). Even so, including random polymer loss or gain, for instance during protocell division, would be an extra source of stochasticity, which would reinforce the conclusion that copying needs to be stronger than random polymerisation to decrease the noise in the system. It would therefore be even more important to reliably produce catalysts for CO 2 fixation, as these could be lost to the environment at any point. The same applies to error rates in either copying or translation, which again we did not include here (this will be the subject of a follow-up paper). Including error would likewise increase the noise in the system, placing an even greater premium on catalysts of copying rather than random polymerisation. It is worth noting that the system outlined here, based on direct physical interactions, is relatively robust to errors, as a hydrophobic amino acid is more likely to be replaced by an equivalent hydrophobic amino acid than its polar opposite, which accounts simply for the apparent optimization of the genetic code ( 64 ). If genetic heredity began with biases in peptide populations within protocells, the results presented here clarify the conditions under which nucleotide polymers could have evolved robustly. A critical requirement is that copying must outcompete random polymerisation, so that useful sequences are maintained rather than lost to noise. The model also reveals the order in which key functions are likely to have emerged. Sequences enabling CO₂ fixation, which amplify monomer supply and drive protocell growth, must have preceded those catalysing templated polymerisation. This conclusion is grounded in the assumption that protocells were autotrophic, synthesising their own monomers, as this links catalysis of growth to increased ability to explore sequence space. On this robust path to genetic heredity, growth came first – setting the stage for the evolution of genetic information. Conflict of Interest Authors declare no conflicts of interest. Author Contributions RNP, MC, AP & NL conceived the ideas and designed the methodology; RNP undertook the computational modelling; RNP, AP & NL analysed the results; RNP, AP & NL led the writing of the manuscript. All authors contributed critically to the drafts and gave final approval for publication. Code https://github.com/raquelnpalmeira/first_growth_then_information Acknowledgements We thank Stuart Harrison and Aaron Halpern for discussions about the origin of heredity that were valuable background to the work presented here. RNP, MC, AP and NL are supported by funding from the Biotechnology and Biological Sciences Research Council BB/V003542/1). POM is supported by funding from the Engineering and Physical Sciences Research Council (EP/X041921/1) and Natural Environment Research Council (NE/X009734/1). NL is supported by funding from the Gates Foundation. References 1. ↵ Moody ERR , Álvarez-Carretero S , Mahendrarajah TA , Clark JW , Betts HC , Dombrowski N , et al. The nature of the last universal common ancestor and its impact on the early Earth system . Nature Ecology & Evolution . 2024 . 2. ↵ Crapitto AJ , Campbell A , Harris AJ , Goldman AD . A consensus view of the proteome of the last universal common ancestor . Ecology and Evolution . 2022 ; 12 ( 6 ): e8930 . OpenUrl 3. ↵ Ouzounis CA , Kunin V , Darzentas N , Goldovsky L . A minimal estimate for the gene content of the last universal common ancestor—exobiology from a terrestrial perspective . Research in Microbiology . 2006 ; 157 ( 1 ): 57 – 68 . OpenUrl CrossRef PubMed 4. ↵ Koonin EV . Comparative genomics, minimal gene-sets and the last universal common ancestor . Nat Rev Microbiol . 2003 ; 1 ( 2 ): 127 – 36 . OpenUrl CrossRef PubMed Web of Science 5. ↵ Weiss MC , Preiner M , Xavier JC , Zimorski V , Martin WF . The last universal common ancestor between ancient Earth chemistry and the onset of genetics . PLoS Genet . 2018 ; 14 ( 8 ): e1007518 . OpenUrl CrossRef PubMed 6. ↵ Orgel LE . Molecular replication . Nature . 1992 ; 358 ( 6383 ): 203 – 9 . OpenUrl CrossRef PubMed 7. ↵ Eigen M . Selforganization of matter and the evolution of biological macromolecules . Naturwissenschaften . 1971 ; 58 ( 10 ): 465 – 523 . OpenUrl CrossRef PubMed Web of Science 8. ↵ Maynard Smith J , Szathmáry E . The Major Transitions in Evolution . Oxford : Oxford University Press ; 1995 . 9. ↵ Spiegelman S. Extracellular evolution of replicating molecules . The neurosciences The Rockefeller University , New York, NY . 1970 : 927 . 10. ↵ Eigen M , Schuster P . The hypercycle. A principle of natural self-organization. Part A: Emergence of the hypercycle . Naturwissenschaften . 1977 ; 64 ( 11 ): 541 – 65 . OpenUrl CrossRef PubMed Web of Science 11. Michod RE . Population biology of the first replicators: on the origin of the genotype, phenotype and organism . American Zoologist . 2015 ; 23 ( 1 ): 5 – 14 . OpenUrl 12. Boerlijst MC , Hogeweg P . Spiral wave structure in pre-biotic evolution: Hypercycles stable against parasites . Physica D: Nonlinear Phenomena . 1991 ; 48 ( 1 ): 17 – 28 . OpenUrl 13. ↵ Zachar I , Fedor A , Szathmáry E . Two different template replicators coexisting in the same protocell: stochastic simulation of an extended chemoton model . PLOS ONE . 2011 ; 6 ( 7 ): e21380 . OpenUrl CrossRef PubMed 14. ↵ Szathmáry E , Demeter L . Group selection of early replicators and the origin of life . J Theor Biol . 1987 ; 128 ( 4 ): 463 – 86 . OpenUrl CrossRef PubMed Web of Science 15. ↵ Lewontin RC . The Units of Selection . Annual Review of Ecology and Systematics . 1970 ; 1 : 1 – 18 . OpenUrl CrossRef 16. ↵ Taylor FJ , Coates D . The code within the codons . Biosystems . 1989 ; 22 ( 3 ): 177 – 87 . OpenUrl CrossRef PubMed Web of Science 17. ↵ Harrison SA , Palmeira RN , Halpern A , Lane N . A biophysical basis for the emergence of the genetic code in protocells . Biochimica et Biophysica Acta (BBA) - Bioenergetics . 2022 : 148597 . 18. ↵ Dill KA . Theory for the folding and stability of globular proteins . Biochemistry . 1985 ; 24 ( 6 ): 1501 – 9 . OpenUrl CrossRef PubMed Web of Science 19. Dill KA , MacCallum JL . The Protein-Folding Problem, 50 Years On . Science . 2012 ; 338 ( 6110 ): 1042 – 6 . OpenUrl Abstract / FREE Full Text 20. ↵ Schwartz R , King J . Frequencies of hydrophobic and hydrophilic runs and alternations in proteins of known structure . Protein Science . 2006 ; 15 ( 1 ): 102 – 12 . OpenUrl CrossRef PubMed Web of Science 21. ↵ Pecoraro VL Moser CC , Sheehan MM , Ennist NM , Kodali G , Bialas C , Englander MT , et al. Chapter Sixteen - De Novo Construction of Redox Active Proteins . In: Pecoraro VL , editor. Methods in Enzymology . 580: Academic Press ; 2016 . p. 365 - 88 . 22. ↵ Harrison SA , Rammu H , Liu F , Halpern A , Palmeira RN , Lane N . Life as a guide to its own origins. Annual Review of Ecology , Evolution, and Systematics . 2023 ; 54 ( 1 ): 327 – 50 . OpenUrl 23. ↵ Batra VK , Beard WA , Shock DD , Krahn JM , Pedersen LC , Wilson SH . Magnesium-induced assembly of a complete DNA polymerase catalytic complex . Structure . 2006 ; 14 ( 4 ): 757 – 66 . OpenUrl CrossRef PubMed 24. A Darst S . Bacterial RNA polymerase . Current Opinion in Structural Biology . 2001 ; 11 ( 2 ): 155 – 62 . OpenUrl CrossRef PubMed Web of Science 25. ↵ Cramer P . RNA polymerase II structure: from core to functional complexes . Current Opinion in Genetics & Development . 2004 ; 14 ( 2 ): 218 – 26 . OpenUrl PubMed 26. ↵ Weber AL , Lacey JC . Genetic code correlations: Amino acids and their anticodon nucleotides . Journal of Molecular Evolution . 1978 ; 11 ( 3 ): 199 – 210 . OpenUrl CrossRef PubMed Web of Science 27. ↵ Buckel W , Thauer RK . Energy conservation via electron bifurcating ferredoxin reduction and proton/Na+ translocating ferredoxin oxidation . Biochimica et Biophysica Acta (BBA)-Bioenergetics . 2013 ; 1827 ( 2 ): 94 – 113 . OpenUrl 28. ↵ Thauer RK , Kaster A-K , Seedorf H , Buckel W , Hedderich R . Methanogenic archaea: ecologically relevant differences in energy conservation . Nature Reviews Microbiology . 2008 ; 6 ( 8 ): 579 – 91 . OpenUrl CrossRef PubMed Web of Science 29. ↵ Nudler E . RNA polymerase active center: the molecular engine of transcription . Annu Rev Biochem . 2009 ; 78 : 335 – 61 . OpenUrl CrossRef PubMed Web of Science 30. Yang W , Lee JY , Nowotny M . Making and breaking nucleic acids: two-Mg2+-ion catalysis and substrate specificity . Molecular Cell . 2006 ; 22 ( 1 ): 5 – 13 . OpenUrl CrossRef PubMed Web of Science 31. ↵ Gesteland RF . Unfolding of Escherichia coli ribosomes by removal of magnesium . Journal of molecular biology . 1966 ; 18 ( 2 ): 356 -IN14. OpenUrl CrossRef PubMed Web of Science 32. ↵ Moran PAP . Random processes in genetics . Mathematical Proceedings of the Cambridge Philosophical Society . 1958 ; 54 ( 1 ): 60 – 71 . OpenUrl CrossRef 33. ↵ Braakman R , Smith E . The emergence and early evolution of biological carbon-fixation . PLOS Computational Biology . 2012 ; 8 ( 4 ): e1002455 . OpenUrl 34. ↵ Muchowska KB , Varma SJ , Moran J . Nonenzymatic metabolic reactions and life’s origins . Chemical Reviews . 2020 ; 120 ( 15 ): 7708 – 44 . OpenUrl CrossRef PubMed 35. Muchowska KB , Varma SJ , Moran J . Synthesis and breakdown of universal metabolic precursors promoted by iron . Nature . 2019 ; 569 ( 7754 ): 104 – 7 . OpenUrl CrossRef PubMed 36. Piedrafita G , Varma SJ , Castro C , Messner CB , Szyrwiel L , Griffin JL , et al. Cysteine and iron accelerate the formation of ribose-5-phosphate, providing insights into the evolutionary origins of the metabolic network structure . PLoS biology . 2021 ; 19 ( 12 ): e3001468 . OpenUrl PubMed 37. Preiner M , Igarashi K , Muchowska KB , Yu M , Varma SJ , Kleinermanns K , et al. A hydrogen-dependent geochemical analogue of primordial carbon and energy metabolism . Nature ecology & evolution . 2020 ; 4 ( 4 ): 534 – 42 . OpenUrl PubMed 38. Varma S , Muchowska K , Chatelain P , Moran J. Native iron reduces CO2 to intermediates and end-products of the acetyl-CoA pathway . Nat Ecol Evol 2 : 1019 – 1024 . 2018 . OpenUrl PubMed 39. Camprubi E , Harrison S , Jordan S , Bonnel J , Pinna S , Lane N . Do soluble phosphates direct the formose reaction towards pentose sugars? Astrobiology . 2022 (ahead of print). 40. ↵ Kopetzki D , Antonietti M . Hydrothermal formose reaction . New Journal of Chemistry . 2011 ; 35 ( 9 ): 1787 – 94 . OpenUrl CrossRef 41. ↵ Crow JF , Kimura M . An Introduction to Population Genetics Theory : Blackburn Press ; 2009 . 42. ↵ DeGuzman V , Vercoutere W , Shenasa H , Deamer D . Generation of oligonucleotides under hydrothermal conditions by non-enzymatic polymerization . Journal of Molecular Evolution . 2014 ; 78 : 251 – 62 . OpenUrl CrossRef PubMed 43. ↵ Monnard P-A , Kanavarioti A , Deamer DW . Eutectic phase polymerization of activated ribonucleotide mixtures yields quasi-equimolar incorporation of purine and pyrimidine nucleobases . Journal of the American Chemical Society . 2003 ; 125 ( 45 ): 13734 – 40 . OpenUrl CrossRef PubMed Web of Science 44. ↵ Mast CB , Schink S , Gerland U , Braun D . Escalation of polymerization in a thermal gradient . Proceedings of the National Academy of Sciences . 2013 ; 110 ( 20 ): 8030 – 5 . OpenUrl Abstract / FREE Full Text 45. ↵ Jerome CA , Kim HJ , Mojzsis SJ , Benner SA , Biondi E . Catalytic synthesis of polyribonucleic acid on prebiotic rock glasses . Astrobiology . 2022 ; 22 ( 6 ): 629 – 36 . OpenUrl CrossRef PubMed 46. ↵ Pinna S , Kunz C , Halpern A , Harrison SA , Jordan SF , Ward J , et al. A prebiotic basis for ATP as the universal energy currency . PLOS Biology . 2022 ; 20 ( 10 ): e3001437 . OpenUrl PubMed 47. ↵ Wimmer JLE , Kleinermanns K , Martin WF . Pyrophosphate and irreversibility in evolution, or why PP(i) is not an energy currency and why nature chose triphosphates . Front Microbiol . 2021 ; 12 : 759359 . OpenUrl PubMed 48. ↵ Sawai H . Catalysis of internucleotide bond formation by divalent metal ions . Journal of the American Chemical Society . 1976 ; 98 ( 22 ): 7037 – 9 . OpenUrl CrossRef PubMed Web of Science 49. ↵ Saenger W Saenger W . Forces Stabilizing Associations Between Bases: Hydrogen Bonding and Base Stacking . In: Saenger W , editor. Principles of Nucleic Acid Structure . New York, NY : Springer New York ; 1984 . p. 116 - 58 . 50. ↵ Kool ET . Hydrogen bonding, base stacking, and steric effects in dna replication . Annu Rev Biophys Biomol Struct . 2001 ; 30 : 1 – 22 . OpenUrl CrossRef PubMed Web of Science 51. ↵ Nunes Palmeira R , Colnaghi M , Harrison SA , Pomiankowski A , Lane N . The limits of metabolic heredity in protocells . Proceedings of the Royal Society B: Biological Sciences . 2022 ; 289 ( 1986 ): 20221469 . OpenUrl PubMed 52. Raffaelli N. Nicotinamide coenzymes synthesis: a case of ribonucleotide emergence or a byproduct of the RNA world? In: Egel R , Lankenau D-H , Mulkidjanian AY , editors. Origins of Life: The Primal Self-Organization . London : Springer ; 2011 . p. 185 - 208 . 53. ↵ White HB . Coenzymes as fossils of an earlier metabolic state . Journal of Molecular Evolution . 1976 ; 7 ( 2 ): 101 – 4 . OpenUrl CrossRef PubMed Web of Science 54. ↵ Fuchs G . Alternative Pathways of Carbon Dioxide Fixation: Insights into the Early Evolution of Life? Annual Review of Microbiology . 2011 ; 65 (Volume 65, 2011): 631 – 58 . OpenUrl CrossRef PubMed Web of Science 55. ↵ Decker K , Jungermann K , Thauer RK . Energy Production in Anaerobic Organisms . Angewandte Chemie International Edition in English . 1970 ; 9 ( 2 ): 138 – 58 . OpenUrl CrossRef PubMed Web of Science 56. ↵ Schoelmerich MC , Müller V . Energy-converting hydrogenases: the link between H2 metabolism and energy conservation . Cellular and Molecular Life Sciences . 2020 ; 77 ( 8 ): 1461 – 81 . OpenUrl CrossRef PubMed 57. ↵ Tran DB , To TH , Tran PD . Mo- and W-molecular catalysts for the H2 evolution, CO2 reduction and N2 fixation . Coordination Chemistry Reviews . 2022 ; 457 : 214400 . OpenUrl 58. ↵ Ragsdale SW , Kumar M . Nickel-Containing Carbon Monoxide Dehydrogenase/Acetyl-CoA Synthase . Chemical Reviews . 1996 ; 96 ( 7 ): 2515 – 40 . OpenUrl CrossRef PubMed Web of Science 59. ↵ Martin W , Baross J , Kelley D , Russell MJ . Hydrothermal vents and the origin of life . Nature Reviews Microbiology . 2008 ; 6 ( 11 ): 805 – 14 . OpenUrl CrossRef PubMed Web of Science 60. ↵ Martin W , Russell MJ . On the origin of biochemistry at an alkaline hydrothermal vent . Philosophical Transactions of the Royal Society B: Biological Sciences . 2007 ; 362 ( 1486 ): 1887 – 926 . OpenUrl CrossRef PubMed 61. ↵ Uchida K , Miwa M . Poly(ADP-ribose) polymerase: structural conservation among different classes of animals and its implications . Mol Cell Biochem . 1994 ; 138 ( 1-2 ): 25 – 32 . OpenUrl CrossRef PubMed Web of Science 62. ↵ Koonin EV , Novozhilov AS . Origin and evolution of the genetic code: the universal enigma . IUBMB Life . 2009 ; 61 ( 2 ): 99 – 111 . OpenUrl CrossRef PubMed Web of Science 63. ↵ Yang NJ , Hinner MJ . Getting across the cell membrane: an overview for small molecules, peptides, and proteins . Methods Mol Biol . 2015 ; 1266 : 29 – 53 . OpenUrl CrossRef PubMed 64. ↵ Freeland SJ , Hurst LD . The genetic code is one in a million . J Mol Evol . 1998 ; 47 ( 3 ): 238 – 48 . OpenUrl CrossRef PubMed Web of Science View the discussion thread. Back to top Previous Next Posted November 17, 2025. Download PDF Email Thank you for your interest in spreading the word about bioRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following First growth, then information: the path to genetic heredity in protocells Message Subject (Your Name) has forwarded a page to you from bioRxiv Message Body (Your Name) thought you would like to see this page from the bioRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share First growth, then information: the path to genetic heredity in protocells Raquel Nunes Palmeira , Marco Colnaghi , Andrew Pomiankowski , Nick Lane bioRxiv 2025.11.17.688785; doi: https://doi.org/10.1101/2025.11.17.688785 Share This Article: Copy Citation Tools First growth, then information: the path to genetic heredity in protocells Raquel Nunes Palmeira , Marco Colnaghi , Andrew Pomiankowski , Nick Lane bioRxiv 2025.11.17.688785; doi: https://doi.org/10.1101/2025.11.17.688785 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Evolutionary Biology Subject Areas All Articles Animal Behavior and Cognition (7629) Biochemistry (17660) Bioengineering (13881) Bioinformatics (41912) Biophysics (21436) Cancer Biology (18578) Cell Biology (25482) Clinical Trials (138) Developmental Biology (13372) Ecology (19889) Epidemiology (2067) Evolutionary Biology (24302) Genetics (15599) Genomics (22483) Immunology (17728) Microbiology (40365) Molecular Biology (17163) Neuroscience (88540) Paleontology (666) Pathology (2830) Pharmacology and Toxicology (4821) Physiology (7637) Plant Biology (15130) Scientific Communication and Education (2045) Synthetic Biology (4290) Systems Biology (9818) Zoology (2269)

Text is read by the "Ask this paper" AI Q&A widget below. Extraction quality varies by source — PMC NXML preserves structure cleanly, OA-HTML may include some navigation residue, and OA-PDF can have broken hyphenation. The publisher copy (via DOI) is the canonical version.

My notes (saved in your browser only)

Ask this paper AI returns verbatim quotes from the full text · source: preprint-html

Answers must be backed by verbatim quotes from this paper's full text. Hallucinated quotes are dropped automatically; if no verbatim passage answers the question, we say so. How this works

Citation neighborhood (no data yet)

We don't have any in-corpus citations linked to this paper yet. This is a recent paper (2025) — citers typically take a year or two to land, and the OpenAlex reference graph may still be filling in.

Source provenance

europepmc
last seen: 2026-05-20T01:45:00.602351+00:00