Inferential planning in the frontal cortex
preprint
OA: gold
CC-BY-4.0
Abstract
How the brain plans and maintains sequences of future actions remains a central question in systems neuroscience. Recent studies in the frontal cortex have revealed that multiple elements of a sequence are represented simultaneously in separable neural subspaces, challenging classical serial models of sequential planning. Here, we show that these representations emerge naturally under inferential planning in which sequential actions are inferred from sensory evidence and goals. Using a hierarchical generative model, we reproduce key neural phenomena observed in primate frontal cortex, including the simultaneous activation of multiple plan elements, the emergence of (almost) orthogonal ‘memory’ subspaces, and their reuse across forward and backward sequence tasks. Our approach provides a mechanistic account of how probabilistic inference over control states gives rise to distributed and dynamic neural representations of plans. This framework not only unifies previously disparate findings on planning, working memory, and motor preparation, but also generates novel, testable predictions about the dynamics of active inference, the role of sensory subspaces, and the impact of uncertainty on sequence processing.
Full text
85,112 characters
· extracted from
preprint-html
· click to expand
Inferential planning in the frontal cortex | bioRxiv /* */ /* */ <!-- <!-- /*! * yepnope1.5.4 * (c) WTFPL, GPLv2 */ (function(a,b,c){function d(a){return"[object Function]"==o.call(a)}function e(a){return"string"==typeof a}function f(){}function g(a){return!a||"loaded"==a||"complete"==a||"uninitialized"==a}function h(){var a=p.shift();q=1,a?a.t?m(function(){("c"==a.t?B.injectCss:B.injectJs)(a.s,0,a.a,a.x,a.e,1)},0):(a(),h()):q=0}function i(a,c,d,e,f,i,j){function k(b){if(!o&&g(l.readyState)&&(u.r=o=1,!q&&h(),l.onload=l.onreadystatechange=null,b)){"img"!=a&&m(function(){t.removeChild(l)},50);for(var d in y[c])y[c].hasOwnProperty(d)&&y[c][d].onload()}}var j=j||B.errorTimeout,l=b.createElement(a),o=0,r=0,u={t:d,s:c,e:f,a:i,x:j};1===y[c]&&(r=1,y[c]=[]),"object"==a?l.data=c:(l.src=c,l.type=a),l.width=l.height="0",l.onerror=l.onload=l.onreadystatechange=function(){k.call(this,r)},p.splice(e,0,u),"img"!=a&&(r||2===y[c]?(t.insertBefore(l,s?null:n),m(k,j)):y[c].push(l))}function j(a,b,c,d,f){return q=0,b=b||"j",e(a)?i("c"==b?v:u,a,b,this.i++,c,d,f):(p.splice(this.i++,0,a),1==p.length&&h()),this}function k(){var a=B;return a.loader={load:j,i:0},a}var l=b.documentElement,m=a.setTimeout,n=b.getElementsByTagName("script")[0],o={}.toString,p=[],q=0,r="MozAppearance"in l.style,s=r&&!!b.createRange().compareNode,t=s?l:n.parentNode,l=a.opera&&"[object Opera]"==o.call(a.opera),l=!!b.attachEvent&&!l,u=r?"object":l?"script":"img",v=l?"script":u,w=Array.isArray||function(a){return"[object Array]"==o.call(a)},x=[],y={},z={timeout:function(a,b){return b.length&&(a.timeout=b[0]),a}},A,B;B=function(a){function b(a){var a=a.split("!"),b=x.length,c=a.pop(),d=a.length,c={url:c,origUrl:c,prefixes:a},e,f,g;for(f=0;f<d;f++)g=a[f].split("="),(e=z[g.shift()])&&(c=e(c,g));for(f=0;f<b;f++)c=x[f](c);return c}function g(a,e,f,g,h){var i=b(a),j=i.autoCallback;i.url.split(".").pop().split("?").shift(),i.bypass||(e&&(e=d(e)?e:e[a]||e[g]||e[a.split("/").pop().split("?")[0]]),i.instead?i.instead(a,e,f,g,h):(y[i.url]?i.noexec=!0:y[i.url]=1,f.load(i.url,i.forceCSS||!i.forceJS&&"css"==i.url.split(".").pop().split("?").shift()?"c":c,i.noexec,i.attrs,i.timeout),(d(e)||d(j))&&f.load(function(){k(),e&&e(i.origUrl,h,g),j&&j(i.origUrl,h,g),y[i.url]=2})))}function h(a,b){function c(a,c){if(a){if(e(a))c||(j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}),g(a,j,b,0,h);else if(Object(a)===a)for(n in m=function(){var b=0,c;for(c in a)a.hasOwnProperty(c)&&b++;return b}(),a)a.hasOwnProperty(n)&&(!c&&!--m&&(d(j)?j=function(){var a=[].slice.call(arguments);k.apply(this,a),l()}:j[n]=function(a){return function(){var b=[].slice.call(arguments);a&&a.apply(this,b),l()}}(k[n])),g(a[n],j,b,n,h))}else!c&&l()}var h=!!a.test,i=a.load||a.both,j=a.callback||f,k=j,l=a.complete||f,m,n;c(h?a.yep:a.nope,!!i),i&&c(i)}var i,j,l=this.yepnope.loader;if(e(a))g(a,0,l,0);else if(w(a))for(i=0;i (function(w,d,s,l,i){w[l]=w[l]||[];w[l].push({'gtm.start':new Date().getTime(),event:'gtm.js'});var f=d.getElementsByTagName(s)[0];var j=d.createElement(s);var dl=l!='dataLayer'?'&l='+l:'';j.src='//www.googletagmanager.com/gtm.js?id='+i+dl;j.type='text/javascript';j.async=true;f.parentNode.insertBefore(j,f);})(window,document,'script','dataLayer','GTM-M677548'); Skip to main content Home About Submit ALERTS / RSS Search for this keyword Advanced Search New Results Inferential planning in the frontal cortex View ORCID Profile Francesco Donnarumma , View ORCID Profile Thomas Parr , View ORCID Profile Karl Friston , View ORCID Profile James Whittington , View ORCID Profile Giovanni Pezzulo doi: https://doi.org/10.1101/2025.11.26.690672 Francesco Donnarumma 1 Institute of Cognitive Sciences and Technologies, National Research Council , Rome, Italy Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Francesco Donnarumma Thomas Parr 2 Nuffield Department of Clinical Neurosciences, University of Oxford , Oxford OX1 2JD, UK Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Thomas Parr Karl Friston 3 Queen Square Institute of Neurology, University College London , London WC1E 6BT, UK 4 VERSES Research Lab , Los Angeles, CA 90016, USA Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Karl Friston James Whittington 5 Department of Experimental Psychology, University of Oxford , Oxford, UK Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for James Whittington Giovanni Pezzulo 1 Institute of Cognitive Sciences and Technologies, National Research Council , Rome, Italy Find this author on Google Scholar Find this author on PubMed Search for this author on this site ORCID record for Giovanni Pezzulo For correspondence: giovanni.pezzulo{at}istc.cnr.it Abstract Full Text Info/History Metrics Preview PDF Abstract How the brain plans and maintains sequences of future actions remains a central question in systems neuroscience. Recent studies in the frontal cortex have revealed that multiple elements of a sequence are represented simultaneously in separable neural subspaces, challenging classical serial models of sequential planning. Here, we show that these representations emerge naturally under inferential planning in which sequential actions are inferred from sensory evidence and goals. Using a hierarchical generative model, we reproduce key neural phenomena observed in primate frontal cortex, including the simultaneous activation of multiple plan elements, the emergence of (almost) orthogonal ‘memory’ subspaces, and their reuse across forward and backward sequence tasks. Our approach provides a mechanistic account of how probabilistic inference over control states gives rise to distributed and dynamic neural representations of plans. This framework not only unifies previously disparate findings on planning, working memory, and motor preparation, but also generates novel, testable predictions about the dynamics of active inference, the role of sensory subspaces, and the impact of uncertainty on sequence processing. 1 Introduction The ability to plan is central to intelligent behaviour. Planning involves sequencing a series of actions that can later be enacted in the world. While we are beginning to understand sequence processing in the brain mechanistically ( Mattar & Lengyel, 2022 ; Miller et al., 2017 ; Pezzulo et al., 2019 ; Balaguer et al., 2016 ), our understanding of how plans are formed, and their relationship to sequence processing more broadly, remains incomplete. The dominant view of planning states that elements of the plan are sequentially sampled one after another, i.e., the plan is generated in series with the neural population representing just one action at any time-point. This view is consistent with many existing models for sequence understanding, as well as planning, such as recurrent neural networks ( Elman, 1990 ; Botvinick & Plaut, 2006 ; Maass et al., 2002 ; Ganguli et al., 2008 ; Jensen et al., 2024 ), probabilistic generative models ( Parr et al., 2024 ; Pezzulo et al., 2017 , 2014 ; George et al., 2021 ), or heteroclinic channels ( Rabinovich et al., 2014 ). Representing plans and sequences sequentially has been able to account for a variety of different neural representations, from motor sequences as dynamical systems ( Mante et al., 2013 ; Sussillo et al., 2015 ) to hippocampal replay sampled from a generative model (cognitive map) of the environment ( Schwartenbeck et al., 2023 ; Stoianov et al., 2022 ; Foster, 2017 ) and the mental planning of multiple tasks reported in the orbitofrontal cortex ( Wilson et al., 2014 ; Schuck et al., 2016 ; Van de Maele et al., 2024 ). Further, such formalisms have been combined with reinforcement learning ( Stachenfeld et al., 2017 ; Behrens et al., 2018 ; Baram et al., 2021 ), as well as the probabilistic planning literature ( George et al., 2021 ; Raju et al., 2024 ; Stoianov et al., 2018 ), to understand goal directed behaviour with models predicting a variety of neural phenomena of the hippocampal formation. Despite their differences, all these models generate sequential transitions between elements—whether spatial positions, memories, or actions. At the neural level, this implies that at any specific point in time, the brain represents only one element of the sequence in ongoing neural activity. Recently, however, a different type of representation has been uncovered in prefrontal cortex. Here, all elements of a sequence are represented simultaneously in neural activity at any given time, from path planning tasks ( Mushiake et al., 2006 ; Saito et al., 2005 ; El-Gaby et al., 2024 ) to sequence memory tasks ( Xie et al., 2022 ; Panichello et al., 2024 ; Chen et al., 2024 ) to drawing ( Averbeck et al., 2002 ) and visual working memory tasks ( Liu et al., 2024 ). Each element of the sequence is represented, and ordered, in a distinct set of neural subspaces—termed ‘activation slots’ ( Whittington et al., 2025 ). This allows the entire sequence to be represented simultaneously across the different slots. While this is a different view of sequence representation, it has been shown to have a mathematical equivalence to the sequential view described earlier ( Whittington et al., 2025 ). This raises the tantalising possibility that this new type of representations—slots—can also serve as a neural substrate of planning more broadly. In this paper, we formalise planning as inference over these ordered subspaces. We show that messages are passed from slot to slot in order to infer the correct state and action held at each time-step of the plan. Using this theory we able to explain a variety of neural findings from prefrontal cortex while animals plan: 1) single neurons coding for specific cursor movements at specific time-steps in the plan ( Mushiake et al., 2006 ); 2) neural subspaces holding arbitrary elements in sequence memory task ( Xie et al., 2022 ); 3) showing that the same subspaces are reused in both forward and backward planning tasks ( Chen et al., 2024 ); 4) inferring plans when the number of sequence elements in a plan differs ( Chen et al., 2024 ). Our theory shows planning and sequence memory to be two sides of the same coin and offers a mechanistic understanding of how an entire plan can be encoded—in neural activity—simultaneously in prefrontal cortex. 2 Results We first introduce an inferential planning model ( Section 2.1 ) used subsequently to reproduce the neural dynamics observed empirically in three studies ( Mushiake et al., 2006 ; Xie et al., 2022 ; Chen et al., 2024 ). 2.1 A prefrontal model to infer arbitrary plans We start by considering what it would mean to represent an arbitrary plan simultaneously in a set of neurons (or neuronal populations), and then construct a model that is able to infer such a plan representation. A plan is a proposed sequence of actions ( a t ) and states ( s t ): { ( s t , a t )}. Actions, however, can be abstract or higher-level subgoals (e.g., go to A) that are subsequently enacted by motor controls (e.g., move arm left). To account for this, we use a hierarchical model, in which the plan is represented in variables abstracted from motor commands (in Level 2), while the motor commands themselves are represented in Level 1 ( Figure 1A ). For Level 2 neurons to represent arbitrary full plans (complete sequences of action and states), multiple copies of action and state neurons are required in which each copy corresponds to potential actions and states at each time-point in the plan ( Figure 1B ). Correspondingly, this means the whole plan is simultaneously encoded by neural activity. For example, if each planning problem involves a sequence of three joystick movements—example sequences being { Up, Left, Left } or { Up, Right, Right } (with corresponding states at each time-point)—there must be three pools of neurons each able to represent any state and any action at their corresponding time-point ( Figure 1C ). Download figure Open in new tab Figure 1. Schematic illustration of the hierarchical inferential planning model. (A) The model comprises two interconnected levels. Level 2 (planning level) encodes beliefs about plans—abstract sequences of actions (and states) over time. Level 1 (sensorimotor level) encodes beliefs about concrete motor commands (and sensory states), which generate observable outcomes in the environment. Arrows indicate the bidirectional exchange of information: top–down predictions from Level 2 guide action selection at Level 1, while bottom–up sensory evidence updates beliefs at both levels. This architecture supports both forward and backward inference over time, allowing the system to infer, maintain, and flexibly revise entire action sequences. (B) Within Level 2, neural populations encode beliefs about abstract actions Up, Down, Left, and Right joystick movements (and corresponding states, omitted for clarity) at multiple time points; for example, U1 is Up at time point 1, U2 is Up at time point 2, etc. These are maintained and updated in parallel through reciprocal message passing ( Section 4 ). (C) Example of Level 2 neural representations of two plans: { Up, Left, Left } and { Up, Right, Right }. See main text for further explanation. (D) Intuitive understanding of the message passing process in which plans are inferred. First a start and goal state are presented, and then through multiple iterations (boxes from left to right) the full plan is inferred. In this schematic, we show both example action neurons and state neurons of Level 2 (for a simplified task compared to the other panels). See the main text for further explanation. This simultaneous representation of all elements in the plan allows for arbitrary plans to be represented. This is fundamentally different to formalisms where elements of a plan are represented sequentially, one after another, in the same neuron or population. While that setup can represent specific sequences using neural dynamics (provided the appropriate synaptic connections), it does not have the flexibility to represent arbitrary plan sequences. Furthermore, the encoding of the first action cannot be revised or contextualised by the last action, because the first action is encoded in the past, when the last action is represented. The ensuing simultaneous sequence element setup bears a remarkable resemblance to the ‘activation slots’ found in prefrontal cortex ( Whittington et al., 2025 ). However, current models of ‘activation slots’ are limited to understanding just sequential memory tasks. In this paper, we show that the same computational architecture can solve arbitrary planning tasks. In order to form a plan, however, inference is required. This may involve inferring the path between a start and goal state, though more generally it is the process of combining observations with task statistics to infer optimal behaviour (we provide examples of this later in the paper). Regardless of the setting, we employ variational Bayes for planning as inference ( Attias, 2003 ; Botvinick & Toussaint, 2012 ; Friston et al., 2017a ; Parr et al., 2022 ). Thus the higher level (Level 2) encodes beliefs about plans or policies , while the lower level (Level 1) encodes beliefs about concrete sensorimotor actions that implement these plans—by moving the agent from one location to the next—interacting directly with the environment. Information flows bidirectionally between the two levels and with the environment. Predictions about planned sequences flow downward from Level 2 to Level 1, guiding the selection of sensorimotor actions that are executed in the environment. In turn, sensory feedback and action outcomes generate bottom-up signals that update beliefs at Levels 1 and then 2. This bidirectional exchange allows the system to continuously align planned sequences with sensory evidence, ensuring coherent behaviour across hierarchical levels of control. Within each level, neural populations represent probabilistic beliefs about hidden states, and their interactions implement probabilistic belief updating via a message passing method, called belief propagation ( Pearl, 2022 ; George & Hawkins, 2009 ; George et al., 2025 ; Friston et al., 2017a , b ; Parr et al., 2019 ). In sum, we consider a (Level 2) neural population which consists of copies (subspaces) of state and action representations, each able to represent an arbitrary state and action. To infer what action and state should go in each subspace (i.e., infer the plan), we use a neuronally plausible message passing algorithm (see Figure 1D for an intuitive illustration). Critically, the simultaneous representation of all time-steps in the plan is essential to support the reciprocal message passing—i.e.,belief propagation—that implements this sequential planning. See Section 4 for a formal description of the model and implicit message passing. We now show a series of results ( Sections 2.2 , 2.3 , 2.4 ) in which our model recapitulates the prefrontal neural activity of several experimental studies. In each case, we show neural population dynamics corresponding to the inference process of Level 2 abstract plans. Where specified, we also show dynamics corresponding to inference of Level 1 specific actions. 2.2 Planning sequences of cursor movements First we consider a planning task ( Mushiake et al., 2006 ) in which monkeys were trained to control a cursor to navigate a two-dimensional maze, moving from a central position to a cued goal location ( Figure 2A-B ). Their key finding was that, during the preparatory period preceding movement, simultaneous representations of multiple future cursor movements were recorded in distinct neuronal populations within the monkey lateral prefrontal cortex, each selective for a specific movement at a specific time step, e.g., the Right cursor movement at the 1st time step, the Left cursor movement at the 2nd time step, or the Left cursor movement at the 3rd time step ( Figure 2C ). In contrast, neural activity in the primary motor cortex primarily reflected actual arm movements. Download figure Open in new tab Figure 2. Planning a sequence of three cursor movements in a maze. This simulation replicates the experimental setup of ( Mushiake et al., 2006 ). (A-B) Monkeys navigated a 5 × 5 ‘maze’ by controlling a joystick. They could perform five possible cursor actions: Up, Down, Left, Right, and Stay. (C) Schematic illustration of the results reported in ( Mushiake et al., 2006 , Figure 3 ), showing three neurons selective for the Right cursor movement at 1st time step, the Left cursor movement at the 2nd time step, and the Left cursor movement at the 3rd time step. These neurons correspond to R(1), L(2) and L(3) in our simulations. (D-F) and (G-I): two example maze navigation problems, consisting in moving from the start (green square) to the goal state (red circle). Each white cell represents a valid navigable state, whereas black cells indicate blocked states. Walls are depicted as solid lines, and open lines ( doors ) represent allowed transitions. For each configuration, the successful strategy (shortest policy) is highlighted with arrows. (D) In this example maze, the planned policy is Up-Left-Left ( ULL ). (E) Raster plots showing simulated neural activity, with four neurons coding for each of the 5 control states ( Up, Down, Left, Right, Stay ) at Level 2, at each each of the three time steps. Neurons coding for actions at the 3 time steps are color coded in red, yellow and violet, respectively. At the end of the simulation, corresponding to the delay period before execution, the policy ULL is coded by the simultaneous activation of three populations of neurons, U (1), L (2) and L (3). (F) The same results, but plotted as simulated firing rates of neurons associated with different actions. Neurons representing U, D, L and R are coded with circles, squares, diamonds, and triangles, respectively. The stay action is not shown for simplicity. (E) In this example maze, the inferred policy is Up-Right-Right (URR). (H-I) Simulated raster plots and firing rates for the policy URR, following the same format as Panels C-D. See the main text for further explanation. We simulated this task under our hierarchical model ( Figure 1 ). The lower level (Level 1) controls cursor movements, whereas the upper level (Level 2) plans the sequence of cursor movements. Level 2 comprised 25 hidden states representing maze locations (from S 11 to S 55 ) and 5 control states corresponding to possible cursor movements: Up, Down, Left, Right, and Stay. The simulated neural activity reported below corresponds to the inference of these 5 control states across the 3 time steps of the plan, which we compare with the frontal cortical activity reported by ( Mushiake et al., 2006 ). We denote time-steps within the plan (but all still represented simultaneously) with parentheses; for instance, the ‘Up’ action at the first, second, and third time steps is represented as U(1), U(2), and U(3), respectively. We use 4 neurons to represent each hidden state, and thus 60 neurons (4 × 5 × 3) to represent the 5 possible cursor movements across 3 time steps Figure 2D illustrates an example problem in which the agent plans a sequence of three cursor movements to navigate from the start location (green; S 33 ) to the goal location (red; S 21 ). Figure 2D shows a simulated raster plot representing the inferential process for the five possible cursor movements (Up, Down, Left, Right, Stay) across the three time steps. The 15 horizontal lines represent the simulated firing of 15 neural populations (each comprising four neurons) encoding the five cursor movements at three time steps. For instance, U(1) (first line), U(2) (sixth line), and U(3) (eleventh line) correspond to planning the ‘Up’ movement at the first, second, and third time steps, respectively. Cursor movements at different time steps are color-coded: red, yellow, and violet correspond to the first, second, and third time steps, respectively. At the start of the planning process (time 0), the agent maintains a relatively flat belief distribution over the first, second, and third planned cursor movements. Upon presentation of the goal location S 21 , it updates its beliefs about all three movements. During the early stage of planning (approximately 0.2–0.6), parallel competition occurs among the two available cursor movements at time step 1 (U(1) and D(1)), the two most likely movements at time step 2 (L(2) and R(2)), and the three most likely movements at time step 3 (U(3), L(3), and R(3)). By around 0.6, planning is completed, and the three selected cursor movements (U(1), L(2), and L(3)) exhibit simultaneous, sustained activation, which can serve as a prospective working memory for the plan. However, as time passes—and the plan is executed at the lower level—mnemonic representations of early elements become postdictive. The dynamics of the inferential process can also be observed in Figure 2F , which shows the average firing rates of the 12 neural populations involved in cursor movement planning (with ‘Stay’ actions omitted). The figure illustrates that the three selected cursor movements (U(1), L(2), and L(3)) gradually increase their activation over time, with the first planned movement (U(1)) rising fastest, reflecting faster convergence during inference. These dynamics resemble a competitive queuing mechanism, in which the first element of a sequence is activated first, but here this pattern emerges spontaneously during inference. It is also apparent that most of the non-selected cursor actions increase slightly before decreasing, as they lose the competition with alternative plans. Unfeasible actions (e.g., L(1) and R(1)) instead show an immediate decrease in activity. Figures 2G–I illustrate another example problem, in which the agent plans a sequence of three cursor movements from the start location (green; S 33 ) to the goal location (red; S 25 ), along with the corresponding inferential dynamics, using the same format as Figures 2D–F . In summary, these simulations demonstrate how inferential planning dynamics naturally reproduce the simultaneous activation and updating of neural populations encoding cursor movements at different time steps, as observed in the monkey lateral prefrontal cortex ( Mushiake et al., 2006 ). 2.3 Inferring and holding in memory a sequence of saccades Second, we consider a task ( Xie et al., 2022 ) in which monkeys were first shown a sequence of three targets (from six possible targets arranged in a hexagon) and then, after a delay, were required to saccade to the same three targets in order ( Figure 3A-D ). Unlike the previous task, this requires inferring a plan from observations and maintaining it in working memory. A key finding of this study was that, during the delay period, the prefrontal representation decomposed into three distinct and near-orthogonal subspaces for each element (referred to as ‘ranks’) in the sequence ( Figure 3C ). Download figure Open in new tab Figure 3. Inferring and holding in memory a sequence of three saccades. This simulation reproduces the experimental setup of ( Xie et al., 2022 ). (A–B) Monkeys were instructed to perform sequences of three saccades toward six possible targets, denoted as { A – F }. (C) Schematic illustration of the results reported in ( Xie et al., 2022 , Figure 2 ). The figure shows the disentangled neural state space representation projected onto three near-orthogonal 2D, step-specific subspaces, one per panel. Within each panel, Black-connected points indicate samples at the end of the delay arranged in a hexagonal geometry; central points correspond to the beginning of the delay. (D) Example of a trial, in which monkeys maintained a central fixation (green square) while they were shown three consecutive targets, A, B , and C , that they had to fixate in the same order after a delay period. (E-F) Simulated firing rates and spike trains for an example trial, requiring generate sequential saccades to the A, B , and C targets. (G) Simulated population responses for each combination of step and target during the delay period, projected onto three 2D step-specific subspaces, one per panel; note that these subspaces correspond to the rank-subspaces in ( Xie et al., 2022 ), shown in Panel C. The bottom-left plot shows that the three subspaces are oriented in a near-orthogonal manner in neural state space, as indicated by the large principal angles between them. The bottom-right plot shows the cumulative explained variance along the different steps in neural state space, reflecting the symmetric partitioning of information across the three subspaces. See the main text for further explanation. Figure 3E–G shows the results of a simulation of the ( Xie et al., 2022 ) experiment. The generative model’s structure is the same as in the previous simulations, but the states at each level differ. Level 1 controls eye movements and is not the focus of our analysis. Level 2 comprises 6 targets for each of the 3 time steps, corresponding to 18 neural populations in Figure 3E . Targets are labelled A − F , and the number in parentheses denotes the time step. At the beginning of an example trial (‘Pre’ period), the agent maintains a relatively flat belief distribution over the next targets. During the observation periods ‘S1’, ‘S2’, and ‘S3’ the agent observes targets A, B , and C , respectively, and incrementally infers that the correct plan is ABC . Notably, during ‘S1’ it infers A (1) with high probability while simultaneously decreasing the probabilities of A (2) and A (3), since targets cannot be repeated. During ‘S2’ it maintains its belief about A (1) and additionally infers B (2). Finally, at ‘S3’ it infers the complete ABC plan and maintains it throughout the delay period. The corresponding average firing rates for this inferential process are shown in Figure 3F , in the same format as the first simulation. To test whether the neural coding in our simulation reproduces the near-orthogonal rank subspaces reported by ( Xie et al., 2022 ), we applied the same analysis approach as the original empirical study, projecting population activity into three 2D step-specific ‘rank-subspaces’. Figure 3G shows the simulated population responses for each target and time step projected onto these subspaces, with the six target locations colourcoded. The results match very well the empirical data. The hexagonal shape of the subspaces is the same as the empirical data ( Figure 3C ). Furthermore, the large principal angles between the three subspaces, together with the cumulative explained variance across steps in neural state space, recapitulate the empirical findings of ( Xie et al., 2022 ) and indicate that the subspaces are nearly orthogonal. In summary, this simulation demonstrates that inferential planning dynamics reproduce the neural population (information) geometry observed by ( Xie et al., 2022 ) during the delay period, when monkeys maintain a sequential plan for three saccades in working memory. 2.4 Inferring and holding in memory a variable-length sequence of forward or backward saccades Third, we consider an experiment ( Chen et al., 2024 ) with an identical setup to the previous task but extended to three novel conditions ( Figure 4A-B and Figure 5A-B ). In the first (Forward) condition, monkeys observed sequences of variable length —from one to three targets—and therefore had to infer the length of their plan. In the second (Backward) condition, monkeys again observed sequences of variable length but were required to reproduce the plan in the backward direction, that is, to make saccades to the targets in the reverse order of presentation. Finally, in the third (Mixed) condition, monkeys observed sequences of fixed length (two targets) and then received a cue instructing them to reproduce the plan either in the forward direction (the same order of presentation) or in the backward direction (the reverse order). Download figure Open in new tab Figure 4. Inferring and holding in memory a variable-length sequence of saccades, to be executed in either the forward or the backward direction. This simulation reproduces the first two conditions of the experimental setup of ( Chen et al., 2024 ). In (A-B), sequences of variable length (one to three elements) are drawn from six possible target { A − F } (e.g., ABC, AB, ABD, BCA, CB, A, B ). For each block of trials, the instruction is to later execute the trial in the forward or backward direction. The targets are presented at stages S1, S2, and S3 (but S2 and S3 targets can be omitted). During the subsequent delay period, the inferred sequence is maintained in working memory. (C-E) Schematic illustration of the results reported in ( Chen et al., 2024 , Figure 3 ), showing neural state activities corresponding to forward ‘memory’ populations, ‘sensory’ populations, and ‘backward’ memory populations. (F-H) Simulation of the ABC sequence in the Forward condition. (F) Raster plots showing simulated neural activity, with four neurons coding for each of the 6 control states (Targets A − F ) at Level 2, at each each of the three time steps. Neurons coding for control states at the 3 time steps are colour coded in red, yellow and violet, respectively. At the end of the simulation, corresponding to the delay period before execution, the policy ABC is coded by the simultaneous activation of three populations of neurons, A (1), B (2), and C (3). (G) The same results, but plotted as simulated firing rates of neurons associated with different actions. (H) The same results, but plotted as simulated dynamics of the first three principal components (PCs) summarizing the neural state space shown in (C) and corresponding to the ‘memory’ subspaces in the forward case in ( Chen et al., 2024 ). (I-K) Simulation of the ABC sequence in the Backward condition, using the same format as (F-H). These activations summarize neural state space shown in (E) and corresponding to the ‘memory’ subspaces in the backward case in ( Chen et al., 2024 ). (L-N) Simulation of Level 1 activations during the observation of the ABC sequence in both the Forward and Backward conditions. This simulation follows the same format as (F–H) but depicts the activity of hidden states at Level 1 rather than Level 2. These activations summarize the neural state space shown in (D) and corresponding to the ‘sensory’ subspaces in ( Chen et al., 2024 ). See the main text for further explanation. Download figure Open in new tab Figure 5. Inferring and holding in memory a fixed-length sequence of saccades, to be executed in either the forward or backward direction. This simulation reproduces the third condition of the experimental setup of ( Chen et al., 2024 ). (A-B) In this task, the sequence length is fixed to two targets, selected from the set of six possible targets { A − F }. The two targets are presented sequentially during epochs S 1 and S 2, followed by a delay period and an instruction cue indicating whether the sequence should be reproduced in the forward or backward direction. During the subsequent delay period, the inferred plan is maintained in working memory. (C) Schematic illustration of the results reported in ( Chen et al., 2024 , Figure 4 ), showing that the first ‘memory’ subspace (‘memory-1’) in the forward condition could be generalized to the second ‘memory’ subspace (‘memory-2’) in the backward condition, and the second ‘memory’ subspace (‘memory-2’) in the backward condition could be generalized to the first ‘memory’ subspace (‘memory-1’) in the forward condition. The horizontal coloured bars indicate time windows where the decoding performance of an algorithm trained to generalize from forward to backward ‘memory’ subspaces (or vice versa) was significantly higher than the chance level, indicating that the subspaces could be effectively generalized. Training and test epochs for the algorithm are shown in green and orange, respectively. (D-F) An example problem: inferring the AB sequence in the forward condition. (D-E) Raster plots showing simulated neural activity and simulated firing rates associated with inferring the AB plan. (F) Simulated dynamics of the two first principal components during the task. (G-I) Another example problem: inferring the AB sequence in the backward condition. (J-L) Another example problem: BA sequence in the backward condition. See the main text for further details. As in ( Xie et al., 2022 ), the first, second, and third actions in the sequence were represented in separate subspaces—here referred to as ‘memory’ subspaces. Additionally, the dynamics of these subspaces exhibited a gradual build-up of activity at the moment when the animal could infer its position within the sequence. For example, if the animal knew that the sequence had to be executed in the forward direction and observed target A at ‘S1’ it could already infer A(1), even if the actual sequence length was still unknown. This was reflected in a rapid build-up of activity corresponding to A at ‘S1’ ( Figure 4C ). Conversely, if the animal knew that the sequence was to be executed in the backward direction and observed target A, it could not yet infer whether the correct action was A(1), A(2), or A(3), resulting in no such build-up of activity ( Figure 4E ). Furthermore, during observation of the three targets (S1, S2, S3), additional ‘sensory’ subspaces transiently coded for the observed targets ( Figure 4D ). Finally, an analysis of the third (Mixed) condition revealed that monkeys used common subspaces across both forward and backward tasks: the first ‘memory’ subspace (‘memory-1’) in the forward condition could be generalized to the second ‘memory’ subspace (‘memory-2’) in the backward condition, and the second ‘memory’ subspace (‘memory-2’) in the backward condition could be generalized to the first ‘memory’ subspace (‘memory-1’) in the forward condition ( Figure 5C ). To model this task, we used the same hierarchical generative model as in the previous simulation, but with an additional hidden state encoding the agent’s belief about whether the task required a ‘forward’ or ‘backward’ plan execution. In the first two conditions, this belief was known before the target sequence was shown, whereas in the third condition it was inferred when the ‘forward’ or ‘backward’ cue appeared. Figure 4F-N shows the simulation of the first two (Forward and Backward) conditions of the ( Chen et al., 2024 ) study, in which the monkeys know from the start whether the plan will be executed in the forward or backward direction but do not know the sequence length. Figure 4F-H shows the simulation of the Forward condition. The gradual build-up of neural activity associated with the ABC sequence to be executed in the forward direction can be seen at three different levels: in the simulated raster plots ( Figure 4F ), the simulated firing rates ( Figure 4G ), and the simulated ‘memory slots’, which simply correspond to 3D projections of the principal components (PCs) of simulated neural activity during inference ( Figure 4H ). In all cases, the build-up of the first target in the plan (A(1)) is faster than that of the second (B(2)) and third (C(3)) targets, consistent with the experimental data ( Figure 4C ). Figure 4I-K presents the results of the simulation for the Backward condition. The sequence of target presentations ( A, B , and C ) was identical to that in the Forward condition, but the temporal dynamics of activity differed. Following the presentation of the first two targets ( A and B ), the model maintained multiple concurrent hypotheses about their possible positions in the sequence ( A (1), A (2), A (3), B (1), B (2), B (3)), reflecting uncertainty about sequence length. This ambiguity persisted until the third target ( C ) was observed, at which point its identity as the final element ( C (3)) became unambiguous, enabling rapid resolution of uncertainty and selective build-up of activity corresponding to the third step. This pattern parallels the empirical findings ( Figure 4E ). Additionally, Figure 4L-N illustrates the transient activation of Level 1 actions inferred during the observation of the three targets, corresponding to the ‘sensory subspaces’ illustrated in Figure 4D . Within the hierarchical model, these Level 1 activations arise from bottom–up inference of the currently observed targets and provide evidence that updates higher-level (Level 2) beliefs about the sequence. Unlike the sustained Level 2 representations, which encode abstract ‘memory’ subspaces, Level 1 activations are short-lived and confined to the sensory epochs, reflecting the transient message passing required for hierarchical inference. Figure 5D-L shows the simulation of the third condition of the study of ( Chen et al., 2024 ), in which monkeys know from the beginning that the plan consists of two targets but are informed about the forward or backward direction only when the cue is presented. In this case, after observing the two targets, A and B , the model maintains parallel, equally plausible, hypotheses, A (1), A (2), B (1), and B (2). After receiving either a ‘forward’ or ‘backward’ cue, the model can correctly infer the AB ( Figure 5D-F ) or the BA plan ( Figure 5G-I ). After observing the two targets, B and A , and a ‘backward’ cue, the model can correctly infer the AB plan ( Figure 5J-L ). This simulation also demonstrates the reuse of common ‘memory subspaces’ for the first and second targets across both forward and backward tasks observed empirically ( Figure 5C ). For example, the neural population encoding A (1) is the same whether the model has observed targets A and B followed by a ‘forward’ cue ( Figure 5A ), or targets B and A followed by a ‘backward’ cue ( Figure 5G ). In summary, this simulation illustrates how inferential planning dynamics reproduce the near-orthogonal, hexagon-shaped subspaces for sequential targets, which are shared across forward and backward tasks, as reported by ( Chen et al., 2024 ). 3 Discussion In this paper, we have shown that the recently discovered ‘activation slots’ are an effective neural substrate for inferring arbitrary plans. Further, we showed the activity of model neurons during inference match experimentally recorded neurons in a variety of planning and sequence memory tasks ( Mushiake et al., 1991 ; Xie et al., 2022 ; Chen et al., 2024 ). We anticipate our planning as active inference formulation, using message passing between activation slots, will serve as a framework for future fine grained mechanistic understanding of planning in prefrontal cortex (and beyond) at the level of neuron and synapse. Indeed, concurrent modelling work has shown that when RNNs are trained on spatial planning tasks, they learn to plan with activation slots—with plans as fixed points of attractor dynamics—with the synaptic connections between slots mirroring the connectivity structure of the spatial environment ( Jensen et al., 2025 ). This bears a relation to our inferential planning via message passing between slots (where, technically, the attractors are variational free energy minima). Our inferential planning approach, however, not only provides a formalism for arbitrary planning tasks, but also provides a formal answer to a key question about neural representation: why plan elements are coded simultaneously. In inferential planning (e.g., active inference, planning as inference, and related approaches ( Attias, 2003 ; Botvinick & Toussaint, 2012 ; Lázaro-Gredilla et al., 2024 ; Friston et al., 2017a ; Parr et al., 2022 ; Ortega & Braun, 2013 ; Gershman & Beck, 2017 ; Kappen et al., 2012 ; George et al., 2021 ; Pezzulo et al., 2018 ; Levine, 2018 ; Isomura et al., 2022 ; Bastos et al., 2012 )), a plan is generated by inferring the most likely sequence of actions (or policy) leading from initial to goal states. Crucially, this inferential process entails reciprocal message passing among representations of past, present, and future (expected) states, which must therefore be maintained and updated in parallel ( George & Hawkins, 2009 ; George et al., 2025 ; Friston et al., 2017a , b ). The simultaneous coding of plan elements is therefore not a nuance but a necessary prerequisite for the message passing that underlies planning. Other models (beyond the aforementioned RNNs ( Whittington et al., 2025 ; Jensen et al., 2025 )) also represent all elements in a sequence simultaneously. Competitive queuing models ( Bullock, 2004 ) generate serial order via a competitive process in which the most active unit wins the competition and generates the corresponding action; it then inhibits itself, allowing the next most active unit to drive the subsequent action, and so forth. This differs from our framework which can endogenously generate a sequence of plan element activations without requiring a predefined queuing mechanism. Indeed, in Figure 2F,I , the firing rate of the first cursor movement increases faster than that of the second and third ones, reflecting a faster resolution of uncertainty about what to do next. Other models represent multiple ordered elements of a sequence simultaneously along a ‘mental line,’ enabling transitive inference ( Jensen et al., 2015 ; Di Antonio et al., 2024 ; Mannella & Pezzulo, 2025 ). This contrasts with our framework, which uses simultaneous representations of actions and states to construct and update plans, rather than to encode relational order. While building an understanding—using a probabilistic formalism—is one level abstracted from neuron and synapse, it will not only be helpful when interpreting representations from tasks that manipulate transition probabilities between goals or those with cue uncertainty ( Findling et al., 2025 ), but also provides a formal account of any neural representation that corresponds to a variable in the underlying generative model. For example, our probabilistic framing suggests that the ‘sensory subspaces’ observed in ( Chen et al., 2024 ) correspond to Level 1 representations that play a central role in hierarchical inference, rather than serving merely as temporary storage. This could be tested by perturbing these sensory subspaces (e.g., optogenetically), which should disrupt the animal’s ability to correctly infer targets, as shown in other studies using hierarchical inference ( Proietti et al., 2023 ; Van de Maele et al., 2024 ; Donnarumma et al., 2025 ). There are several (addressable) limitations of our model. First, we predefined separate neural populations for each element of a plan, which differs from the frontal cortex observations in ( Xie et al., 2022 ; Panichello et al., 2024 ) in which subspaces are distributed across neurons. Such a distributed coding scheme could, however, be readily recovered by assuming a different factorization of the generative model. Second, for simplicity, we focused only on control states at Level 2 and actions at Level 1, ignoring the dynamics of other latent states. Future work could ask whether incorporating these latent states provides additional insights into sequence processing in the frontal cortex. Third, we only analysed the planning phase, rather than the execution phase. Thus our neural subspaces correspond to distinct (and fixed) positions (e.g., 1,2,3,…) of the plan. This is subtly different to the slots model ( Whittington et al., 2025 ) in which slots correspond to relative positions (e.g., present, past, or future) with slot contents moving between slots so that the present slot always contains the present sequence element. This relative representation makes readout (i.e., execution) extremely efficient as there only needs to be one set of readout weights. While we modelled planning with fixed slots, we posit that, in prefrontal cortex, during the execution phase the contents of slots will shuffle between one-another to take advantage of the simple readout mechanism. Indeed experimental data suggests this, with planning phases using using fixed slots ( Chen et al., 2024 ) (like our model), but execution passing contents between slots ( El-Gaby et al., 2024 ) (like the slots model). Frontal cortex has long been known to be central to higher level cognitive functions including cognitive control ( Miller & Cohen, 2001 ; Duncan, 2001 ; Pezzulo et al., 2018 ; Koechlin et al., 2003 ; Barceló, 2021 ; Proietti et al., 2025 ), cognitive map formation ( Schuck et al., 2016 ; Wang & Hayden, 2021 ; Behrens et al., 2018 ), working memory ( Curtis & D’Esposito, 2003 ) and planning ( Mattar & Lengyel, 2022 ; Koechlin, 2016 ; Goel & Grafman, 1995 ; Duncan et al., 1996 ). In this work we provide a formalism of planning built from recent experimental results for simultaneous sequence element coding in prefrontal cortex. In doing so we provide explanations for several empirical observations, mechanistically link planning and working memory, and generating testable predictions for future experiments. 4 Methods In this section, we first provide a concise formal overview of the active inference framework that we adopt in this study ( Parr et al., 2022 ) ( Section 4.1 ). Next, we describe the message-passing scheme that underlies neural simulations in active inference ( Section 4.2 ) and detail the specific hierarchical generative model used to implement the simulations ( Section 4.3 ). Finally, we outline the analytical procedures employed to characterize low-dimensional neural subspaces in the simulation of the study by ( Xie et al., 2022 ; Chen et al., 2024 ) ( Section 4.4 ). 4.1 Brief introduction to Active Inference Active Inference provides a normative account of perception–action as variational Bayesian inference under a generative model, coupled with policy selection that minimizes expected free energy (EFE) ( Parr et al., 2022 ). An agent maintains a probabilistic model over latent (hidden) states and observations—typically represented as a probabilistic graphical model ( Bishop, 2006 )—and acts to realize outcomes that jointly pursue utility maximization ( pragmatic value ) and uncertainty minimization ( epistemic value or information gain ). In the following, we summarize the key ingredients of active inference, from model formalization to the steps required to perform inference by updating hidden states, policies, and precision. Generative model Let O 0: T = ( O 0 , …, O T ) denote a sequence of T observations, S 0: T a sequence of T hidden states, π = ( π 1 , …, π T ) a policy (a sequence of control states, or more simply “actions”), γ ∈ ℝ + a precision controlling the sharpness (i.e., inverse temperature) of policy selection, and Θ the model parameters. A convenient factorization (here, dealing with a single hierarchical level) is: Parameterization We write Θ = { A, B, C, D, E , β } with: Likelihood A : p ( O t | S t , Θ ) = A . It maps hidden causes S to observations O (often categorical). Transitions B : p ( S t +1 | S t , π t , Θ ) = B ( π t ), i.e., policy-conditioned dynamics. Preferences C : an a priori (log-)distribution over outcomes, encoding what the agent prefers to observe, P ( O τ ) ≡ C . Initial prior D : prior on hidden states p ( S 0 | Θ ) = D . Policy prior E : habitual/structural prior over policies. Precision γ : sampled from a Gamma prior p ( γ | Θ ) ~ Γ(1, β ). Intuitively, A tells the agent ‘what sensations to expect from each state’; B tells it ‘how hidden states change under a chosen policy’; C encodes preferences; D is what it believes before seeing anything; E captures habits. To move to a hierarchical specification, we would condition the D hidden state priors for one level on the hidden states at the level above, such that a hidden state at any given level of the model predicts the initial state of a short sequence at the level below. Approximate posterior and mean-field form A common implementation of active inference uses a tractable variational posterior (see mean-field theory Parisi, 1988 ): With this distribution, we maintain a posterior over the policy Q ( π ) and, for each time step, a posterior over hidden states conditional on the policy. Variational free energy (VFE) Perceptual inference at time t minimizes The first term is the Kullback-Leibler (KL) divergence representing the complexity cost (deviation from prior), the second is the expected negative log-likelihood of the observation (negative accuracy , lack of fit to data). Balancing the two yields predictive beliefs ( Parr et al., 2022 ). Under a standard categorical parameterization, a fixed-point update has the form Here σ (·) is the componentwise Softmax, denotes the posterior state marginals and o t is the one-hot encoding of the observed outcome at time t, Q ( π ) = Cat( π ) is the posterior over π , while Q t ( π t ) and are respectively the posterior and the prior over π t . Intuitively, Eq. (4a) : this combines current sensory evidence with predicted messages that depend upon predictions based on beliefs about previous states and postdictions. Note that in Eq. (4a) , B and B T are both assumed to have normalized (sum-to-one) columns. from beliefs about subsequent states; Eq. (4b) : prefer policies that are habituated (large ln E ) and have low EFE; Eq. (4c) : increase precision when expected free energy is low/consistent. The form of Eq. (4a) depends upon a marginal approximation to the prior for states in Eq. (3) that accounts for beliefs about both past and future states (c.f., Bayesian smoothing). For technical details of this scheme, please see ( Parr et al., 2019 ) and the Appendices of ( Parr et al., 2021 ; Friston et al., 2017a ). Expected free energy (EFE) and action selection Policies are scored by their EFE: where Q ( O τ , S τ | π ) ≜ P ( O τ | S τ ) Q ( S τ | π ) is the predictive posterior distribution, is the predicted outcome, P ( O τ ) ≡ C encodes preferences, and ℍ[·] is Shannon entropy. Intuitively, the first term ( expected cost /risk) favours policies that make preferred outcomes likely; the second ( expected ambiguity ) favours policies that reduce uncertainty by seeking informative states. In this model, action selection corresponds to the generation of predicted control signals, which are sampled from the policy posterior in Eq. (4b) . These samples represent planned actions, reflecting the agent’s internal simulation of future behaviour under each candidate policy ( Parr et al., 2022 ). 4.2 Message Passing and neural architecture In this section, we briefly outline the way in which we can interpret the fixed point scheme outlined about in terms of neuronal message passing, with a focus on Eq. (4a) . The idea is relatively simple. If we assume neuronal membrane potentials for a pool of neurons, on average, are represented by a vector , and firing rates (again, on average) by , then treating the softmax function as a neuronal transfer function, such that , we arrive at an interpretation (up to an additive constant) of . We can then express the dynamics of the membrane potentials to a first order approximation as a simple attractor system whose fixed point corresponds to the solution of Eq. (4a) : Here, the key points to note are that the membrane potentials associated with a given population (indexed by time and policy) that represent beliefs about a particular time evolve such that they depend upon the potentials of populations representing beliefs about the immediate past and future. This implies that at any given time, we need to simultaneously hold beliefs about other times to instantiate these dynamics. It is this distinction between the time at which beliefs are held, and the times those beliefs are about that underwrites the core ideas in this paper. Before we move on, it is worth highlighting that this property is not specific to the message passing scheme outlined here, and would also apply to strict belief-propagation or variational message passing schemes. The reason for this is inherent in the structure of the generative model, rather than of the particular method of solution. Specifically, the model used here relies in part upon a Markov chain, in which the Markov blanket of any given state includes both its predecessor and successor. As such, beliefs about proximal time-points will always be informative about the current time. 4.3 Hierarchical generative model for planning The hierarchical Active Inference model employed in this study is designed to structure motor planning across two hierarchical levels, denoted by l . At each level, the model encodes hidden variables that represent distinct aspects of motor organization: low-level motor actions at Level 1 and structured sequences of these actions at Level 2. The overall architecture of the model is illustrated in Figure 6 . Download figure Open in new tab Figure 6. The hierarchical (deep) Active Inference model for action planning. The model consists of two interacting layers, Level 1 and Level 2, organized within the Active Inference framework. Gray nodes denote hidden states and policies, yellow nodes represent observations, and edges illustrate probabilistic dependencies between variables. At the higher level, policies π encode sequences of control states (abstract actions), whereas at the lower level, they correspond to single motor commands. Rectangular nodes depict probabilistic distributions – the conditional dependencies defining the Hidden Markov Models – parameter-ized by the matrices A, B, C, D , and E . The A matrix encodes the likelihood model (how hidden states S t generate observations O t ); the B matrix parameterizes state transitions under each policy π ; the C matrix specifies preferred outcomes and contributes to the expected free energy G ; the E matrix represents prior preferences over policies (representing habits). Notably, the hidden states at Level 1 serve as observations for the higher-level process at Level 2, linking temporal inference across hierarchical timescales. See the main text for further explanation. To capture the structured dependencies among motor components, the hidden state space at each level l is expressed as a tensor product: where each factor (subscripts here being factor indices, and not time-points as previously) encodes a specific dimension of motor planning and control: Hidden Control States – These variables specify the motor controls of the actions themselves. At the lowest level ( l = 1), they represent discrete atomic actions, while at the higher level ( l = 2) they encode sequences of actions composing a coherent motor strategy. Hidden Location States – This factor encodes the spatial component of the motor plan. At Level 1, it corresponds to concrete movement execution in physical space, whereas at Level 2, it represents the structural organization of sequence of action locations corresponding to the policy. Hidden Context States – This component defines the contextual or temporal structure of the motor sequence. In our simulations, at Level 2 it specifies the order in which actions are to be planned (e.g., forward vs. backward order relative to the presentation). Observations Each observation O = O 1 ⊗ O 2 ⊗ O 3 is defined as the tensor product of the following components: O 1 : the observation of control actions. O 2 : the observation of target locations. O 3 : the cue indicating the execution order, which specifies the sequence in which actions should be performed. The cue O 3 is used in Simulation (4) (see Table 1 ) and includes options such as ‘forward,’ ‘backward,’ other possible execution orders, and ‘idk’—a neutral cue indicating that execution does not depend on the presentation order. View this table: View inline View popup Download powerpoint Table 1. Schematics of the simulation parameters. The table summarizes hierarchical Active Inference settings for four tasks: (1) planning sequences of cursor movements (( Mushiake et al., 2006 ), Fig. 2 ); (2) inferring and holding in memory a sequence of saccades (( Xie et al., 2022 ), Fig. 3 ); (3) inferring and holding in memory a variable-length sequence of forward or backward saccades (( Chen et al., 2024 ), Fig. 4 ); and (4) inferring and holding in memory a fixed-length sequence of saccades executed in either forward or backward order (( Chen et al., 2024 ), Fig. 5 ). For each task, the table reports hidden state configurations, prior preferences, and policy constraints. Simulation (1) uses environment-constrained policies and goal preferences. Simulation (2) expands target states across time steps to encode temporal information, with no contextual cue. Simulation (3) introduces variable-length sequences with contextual cues known from the start (Forward or Backward). Simulation (4) uses fixed-length sequences with preferences for both directions, where the cue is revealed at time step t cue . Transition mapping The transition probability given control states is specified by the tensor: where: governs transitions among control states. defines transitions among spatial locations across time steps. Transitions are nearly deterministic, but stochasticity is introduced to model execution errors or uncertainty in control signals. where ϵ is a small noise term and n is the number of allowable possible locations (where a transition is actually possible). Low-level generative model (Level 1) In simulations of tasks from ( Mushiake et al., 2006 ), represents the hidden controls of the cursor actions (i.e. Up, Down, Left, Right). In simulations of tasks from ( Xie et al., 2022 ) and ( Chen et al., 2024 ), represents the hidden control for the saccadic movements towards the targets A, B, C, D, E, F. In both cases, represents the hidden locations of the target states. Likelihood mapping at Level 1 The likelihood p ( O (1) | S (1) ) between hidden states S (1) and observations O (1) is specified through the tensor: where: maps hidden control states to observations O 1 (primitive control actions). maps hidden target states to observations O 2 (target locations). The mapping is almost deterministic but implements cosine tuning for sensory encoding: where θ O and θ S represent preferred directions of observation and hidden state, respectively. Recognition accuracy is tuned in order to get approximately 85% with respect of the preferred direction, and misclassification occurs toward the closest state in angular space. Higher-level generative model (Level 2) Level 2 comprises three hidden-state factors, analogous to those in Level 1, representing beliefs about Controls and Locations . The third factor, Context , encodes the sequence execution order (e.g., ‘forward’, ‘backward’) or its absence (‘idk’). This contextual factor may be fixed prior to the trial (as in simulations from ( Xie et al., 2022 )) or revealed during the trial through observation of the cue O 3 . Likelihood mapping at Level 2 The likelihood p ( S (1) | S (2) ) specifies how higher-level hidden states S (2) predict lower-level hidden states S (1) . It is represented by the tensor: where: : maps higher-level (sequence of) control states to lower-level controls ; : maps higher-level (sequence of) target states to lower-level locations ; : maps higher-level context states to the execution order cue O 3 . Here, higher-level states are mapped onto lower-level states (or cues in the case of S (2) 3 , which are treated as observations in the hierarchical model. Formally, the likelihood mappings are defined as , and , with each mapping being nearly deterministic while allowing for a small degree of stochasticity (approximately 5% noise) in recognizing the correct correspondence. Observations O 3 (e.g., ‘Forward’ or ‘Backward’ cues) are included in tasks from ( Chen et al., 2024 ) to indicate execution order. Preferences The tensor encodes prior preferences over observations: : preferences over motor actions (goal-directed behaviour). : preferences over spatial positions (preferred locations). In the simulations of tasks from ( Mushiake et al., 2006 ) this preference was used to setup target goal states. : preferences over execution order cue (e.g., ‘Forward’ vs. ‘Backward’). Prior beliefs The tensor encodes priors over hidden states: : prior over Controls (initial policy or actions distribution). : prior over Locations (starting position). : prior over Context (execution order). Additionally, E (2) encodes the prior distribution over policies. In line with the simulations from ( Chen et al., 2024 ), where sequences vary in length, we assign equal prior probability to policies of length 1, 2, and 3. This compensates for the combinatorial bias favoring longer sequences (which are more numerous), ensuring that, for instance, upon presentation of the first target, there remains an equal prior likelihood that the full sequence will be of length 1, 2, or 3. Tab 1 shows a summary of main settings for the tasks simulated in the paper. Note that, for simplicity, the generative models used in this study were predefined, reflecting the fact that monkeys had extensively practiced the tasks before neural recordings. However, in principle, such models could also be learned through interaction with the environment, as demonstrated in previous work ( Friston et al., 2016 , 2024 ; Van de Maele et al., 2024 ; George et al., 2021 ). 4.4 Subspace analysis The procedures for generating the 2D plots follow the approach described in ( Xie et al., 2022 ). Neural responses were obtained by performing a linear regression of spike counts during the planning period against one-hot encoded task vectors (e.g., an 18-dimensional vector representing 6 directions across 3 steps). This yielded, for each neuron, a set of regression coefficients β ( r, l ), where r denotes the step and l the direction. For clarity, we focus on three time steps and six directions, although the procedure generalizes to other configurations (e.g., 6 directions and 2 steps, see Fig. 5 ). To capture variance in neural responses—attributable to item variation at each step—we applied principal component analysis (PCA) to the regression coefficients grouped by step. This analysis produced a reliable state-space representation that captures both the relationships among step-specific subspaces and the geometry of spatial representations within each subspace. A multivariable linear regression model was used to determine how task variables influence the average neural response during the late delay period (1 s before the go signal). For example, a length-3 sequence can be represented as an 18-dimensional three-hot vector. A sequence of targets ECA corresponds to indices [5, 3, 1] and can be encoded as: We defined 18 one-hot vectors S r,l as task variables, where r ∈ {1, 2, 3} and l ∈ {1, …, 6}. The average neural response of the i th neuron in one trial during the late delay was modeled as: where β i ( r, l ) are regression coefficients and ϵ i is trial-by-trial noise. To prevent overfitting, we applied Lasso regularization and selected the regularization amplitude via maximum likelihood. The regression coefficients β ( r, l ) were then used to identify low-dimensional subspaces capturing most task-related variance. Specifically, with N neurons, an N -dimensional vector represents each rank–item combination ( r, l ) at the population level. To capture variance due to item differences at each rank, the 18 vectors β ( r, l ) ( r = 1, 2, 3; l = 1, …, 6) were divided into three groups by rank. For each group (fixed r ), PCA was performed to extract the first two principal components, providing a compact representation of spatial geometry within each step-specific subspace. Acknowledgments This research received funding from the European Research Council under the Grant Agreement No. 820213 (ThinkAhead), the Italian National Recovery and Resilience Plan (NRRP), M4C2, funded by the European Union – NextGenerationEU (Project IR0000011, CUP B51E22000150006, ‘EBRAINS-Italy’; Project PE0000013, ‘FAIR’; Project PE0000006, ‘MNESYS’), and the Ministry of University and Research, PRIN PNRR P20224FESY and PRIN 20229Z7M8N. The GEFORCE Quadro RTX6000 and Titan GPU cards used for this research were donated by the NVIDIA Corporation. T.P. is supported by an NIHR Academic Clinical Fellowship [ref: ACF-2023-13-013]. KF is supported by funding from the Wellcome Trust (Ref: 226793/Z/22/Z). J.C.R.W is supported by European Research Council Starting Grant No. 101222868 (NARFB). We used a Generative AI model to correct typographical errors and edit language for clarity. Funder Information Declared European Research Council , 820213 , 101222868 Wellcome Trust, https://ror.org/029chgv08 , 226793/Z/22/Z NIHR Academic Clinical Fellowship , ACF-2023-13-013 References ↵ Attias , H. ( 2003 ). Planning by probabilistic inference . International workshop on artificial intelligence and statistics , 9 – 16 . ↵ Averbeck , B. B. , Chafee , M. V. , Crowe , D. A. , & Georgopoulos , A. P. ( 2002 ). Parallel processing of serial movements in prefrontal cortex . Proceedings of the National Academy of Sciences , 99 ( 20 ), 13172 – 13177 . OpenUrl Abstract / FREE Full Text ↵ Balaguer , J. , Spiers , H. , Hassabis , D. , & Summerfield , C. ( 2016 ). Neural mechanisms of hierarchical planning in a virtual subway network . Neuron , 90 ( 4 ), 893 – 903 . OpenUrl CrossRef PubMed ↵ Baram , A. B. , Muller , T. H. , Nili , H. , Garvert , M. M. , & Behrens , T. E. J. ( 2021 ). Entorhinal and ventromedial prefrontal cortices abstract and generalize the structure of reinforcement learning problems . Neuron , 109 ( 4 ), 713 – 723 . OpenUrl CrossRef PubMed ↵ Barceló , F. ( 2021 ). A predictive processing account of card sorting: Fast proactive and reactive frontoparietal cortical dynamics during inference and learning of perceptual categories . Journal of Cognitive Neuroscience , 33 ( 9 ), 1636 – 1656 . OpenUrl PubMed ↵ Bastos , A. M. , Usrey , W. M. , Adams , R. A. , Mangun , G. R. , Fries , P. , & Friston , K. J. ( 2012 ). Canonical microcircuits for predictive coding . Neuron , 76 ( 4 ), 695 – 711 . OpenUrl CrossRef PubMed Web of Science ↵ Behrens , T. E. , Muller , T. H. , Whittington , J. C. , Mark , S. , Baram , A. B. , Stachenfeld , K. L. , & Kurth-Nelson , Z. ( 2018 ). What is a cognitive map? organizing knowledge for flexible behavior . Neuron , 100 ( 2 ), 490 – 509 . OpenUrl CrossRef PubMed ↵ Bishop , C. M. ( 2006 ). Pattern recognition and machine learning . Springer . ↵ Botvinick , M. & Toussaint , M. ( 2012 ). Planning as inference . Trends in cognitive sciences , 16 ( 10 ), 485 – 488 . OpenUrl CrossRef PubMed Web of Science ↵ Botvinick , M. M. & Plaut , D. C. ( 2006 ). Short-term memory for serial order: a recurrent neural network model . Psychological review , 113 ( 2 ), 201 . OpenUrl CrossRef PubMed Web of Science ↵ Bullock , D. ( 2004 ). Adaptive neural models of queuing and timing in fluent action . Trends in cognitive sciences , 8 ( 9 ), 426 – 433 . OpenUrl CrossRef PubMed Web of Science ↵ Chen , J. , Zhang , C. , Hu , P. , Min , B. , & Wang , L. ( 2024 ). Flexible control of sequence working memory in the macaque frontal cortex . Neuron , 112 ( 20 ), 3502 – 3514 . OpenUrl CrossRef PubMed ↵ Curtis , C. E. & D’Esposito , M. ( 2003 ). Persistent activity in the prefrontal cortex during working memory . Trends in cognitive sciences , 7 ( 9 ), 415 – 423 . OpenUrl CrossRef PubMed Web of Science ↵ Di Antonio , G. , Raglio , S. , & Mattia , M. ( 2024 ). A geometrical solution underlies general neural principle for serial ordering . Nature Communications , 15 ( 1 ), 8238 . OpenUrl PubMed ↵ Donnarumma , F. , Frosolone , M. , & Pezzulo , G. ( 2025 ). Integrating large language models and active inference to understand eye movements in reading and dyslexia . Physics of Life Reviews , 55 , 61 – 78 . OpenUrl PubMed ↵ Duncan , J. ( 2001 ). An adaptive coding model of neural function in prefrontal cortex . Nature reviews neuroscience , 2 ( 11 ), 820 – 829 . OpenUrl CrossRef PubMed Web of Science ↵ Duncan , J. , Emslie , H. , Williams , P. , Johnson , R. , & Freer , C. ( 1996 ). Intelligence and the frontal lobe: The organization of goal-directed behavior . Cognitive psychology , 30 ( 3 ), 257 – 303 . OpenUrl CrossRef PubMed Web of Science ↵ El-Gaby , M. , Harris , A. L. , Whittington , J. C. , Dorrell , W. , Bhomick , A. , Walton , M. E. , Akam , T. , & Behrens , T. E. ( 2024 ). A cellular basis for mapping behavioural structure . Nature , 1 – 10 . ↵ Elman , J. L. ( 1990 ). Finding structure in time . Cognitive science , 14 ( 2 ), 179 – 211 . OpenUrl CrossRef ↵ Findling , C. , Hubert , F. , Laboratory , I. B. , Acerbi , L. , Benson , B. , Benson , J. , Birman , D. , Bonacchi , N. , Buchanan , E. K. , Bruijns , S. , et al. ( 2025 ). Brain-wide representations of prior information in mouse decision-making . Nature , 645 ( 8079 ), 192 – 200 . OpenUrl CrossRef PubMed ↵ Foster , D. J. ( 2017 ). Replay comes of age . Annual review of neuroscience , 40 ( 1 ), 581 – 602 . OpenUrl CrossRef PubMed ↵ Friston , K. , FitzGerald , T. , Rigoli , F. , Schwartenbeck , P. , & Pezzulo , G. ( 2017a ). Active inference: a process theory . Neural computation , 29 ( 1 ), 1 – 49 . OpenUrl CrossRef PubMed ↵ Friston , K. , FitzGerald , T. , Rigoli , F. , Schwartenbeck , P. , Pezzulo , G. , et al. ( 2016 ). Active inference and learning . Neuroscience & Biobehavioral Reviews , 68 , 862 – 879 . OpenUrl PubMed ↵ Friston , K. J. , Da Costa , L. , Tschantz , A. , Kiefer , A. , Salvatori , T. , Neacsu , V. , Koudahl , M. , Heins , C. , Sajid , N. , Markovic , D. , et al. ( 2024 ). Supervised structure learning . Biological Psychology , 193 , 108891 . OpenUrl PubMed ↵ Friston , K. J. , Parr , T. , & de Vries , B. ( 2017b ). The graphical brain: Belief propagation and active inference . Network neuroscience , 1 ( 4 ), 381 – 414 . OpenUrl PubMed ↵ Ganguli , S. , Huh , D. , & Sompolinsky , H. ( 2008 ). Memory traces in dynamical systems . Proceedings of the national academy of sciences , 105 ( 48 ), 18970 – 18975 . OpenUrl Abstract / FREE Full Text ↵ George , D. & Hawkins , J. ( 2009 ). Towards a mathematical theory of cortical micro-circuits . PLoS computational biology , 5 ( 10 ), e1000532 . OpenUrl ↵ George , D. , Lázaro-Gredilla , M. , Lehrach , W. , Dedieu , A. , Zhou , G. , & Marino , J. ( 2025 ). A detailed theory of thalamic and cortical microcircuits for predictive visual inference . Science Advances , 11 ( 6 ), eadr6698 . OpenUrl CrossRef PubMed ↵ George , D. , Rikhye , R. V. , Gothoskar , N. , Guntupalli , J. S. , Dedieu , A. , & Lázaro-Gredilla , M. ( 2021 ). Clone-structured graph representations enable flexible learning and vicarious evaluation of cognitive maps . Nature communications , 12 ( 1 ), 2392 . OpenUrl PubMed ↵ Gershman , S. J. & Beck , J. M. ( 2017 ). Complex probabilistic inference: From cognition to neural computation . Computational models of brain and behavior , 453 – 466 . ↵ Goel , V. & Grafman , J. ( 1995 ). Are the frontal lobes implicated in “planning” functions? interpreting data from the tower of hanoi . Neuropsychologia , 33 ( 5 ), 623 – 642 . OpenUrl CrossRef PubMed Web of Science ↵ Isomura , T. , Shimazaki , H. , & Friston , K. J. ( 2022 ). Canonical neural networks perform active inference . Communications Biology , 5 ( 1 ), 55 . OpenUrl PubMed ↵ Jensen , G. , Muñoz , F. , Alkan , Y. , Ferrera , V. P. , & Terrace , H. S. ( 2015 ). Implicit value updating explains transitive inference performance: The betasort model . PLoS computational biology , 11 ( 9 ), e1004523 . OpenUrl PubMed ↵ Jensen , K. T. , Doohan , P. , Sablé-Meyer , M. , Reinert , S. , Baram , A. , Akam , T. , & Behrens , T. E. ( 2025 ). A mechanistic theory of planning in prefrontal cortex . bioRxiv . ↵ Jensen , K. T. , Hennequin , G. , & Mattar , M. G. ( 2024 ). A recurrent network model of planning explains hippocampal replay and human behavior . Nature neuroscience , 27 ( 7 ), 1340 – 1348 . OpenUrl CrossRef PubMed ↵ Kappen , H. J. , Gómez , V. , & Opper , M. ( 2012 ). Optimal control as a graphical model inference problem . Machine learning , 87 ( 2 ), 159 – 182 . OpenUrl ↵ Koechlin , E. ( 2016 ). Prefrontal executive function and adaptive behavior in complex environments . Current opinion in neurobiology , 37 , 1 – 6 . OpenUrl CrossRef PubMed ↵ Koechlin , E. , Ody , C. , & Kouneiher , F. ( 2003 ). The architecture of cognitive control in the human prefrontal cortex . Science , 302 ( 5648 ), 1181 – 1185 . OpenUrl Abstract / FREE Full Text ↵ Lázaro-Gredilla , M. , Ku , L. , Murphy , K. P. , & George , D. ( 2024 ). What type of inference is planning? Advances in Neural Information Processing Systems , 37 , 116705 – 116742 . OpenUrl ↵ Levine , S. ( 2018 ). Reinforcement learning and control as probabilistic inference: Tutorial and review . arXiv preprint arxiv: 1805.00909 . ↵ Liu , B. , Alexopoulou , Z.-S. , & van Ede , F. ( 2024 ). Jointly looking to the past and the future in visual working memory . Elife , 12 , RP90874 . OpenUrl CrossRef PubMed ↵ Maass , W. , Natschläger , T. , & Markram , H. ( 2002 ). Real-time computing without stable states: A new framework for neural computation based on perturbations . Neural computation , 14 ( 11 ), 2531 – 2560 . OpenUrl CrossRef PubMed Web of Science ↵ Mannella , F. & Pezzulo , G. ( 2025 ). Transitive inference as probabilistic preference learning . Psychonomic Bulletin & Review , 32 ( 2 ), 674 – 689 . OpenUrl PubMed ↵ Mante , V. , Sussillo , D. , Shenoy , K. V. , & Newsome , W. T. ( 2013 ). Context-dependent computation by recurrent dynamics in prefrontal cortex . nature , 503 ( 7474 ), 78 – 84 . OpenUrl CrossRef PubMed Web of Science ↵ Mattar , M. G. & Lengyel , M. ( 2022 ). Planning in the brain . Neuron , 110 ( 6 ), 914 – 934 . OpenUrl CrossRef PubMed ↵ Miller , E. K. & Cohen , J. D. ( 2001 ). An integrative theory of prefrontal cortex function . Annual review of neuroscience , 24 ( 1 ), 167 – 202 . OpenUrl CrossRef PubMed Web of Science ↵ Miller , K. J. , Botvinick , M. M. , & Brody , C. D. ( 2017 ). Dorsal hippocampus contributes to model-based planning . Nature neuroscience , 20 ( 9 ), 1269 – 1276 . OpenUrl CrossRef PubMed ↵ Mushiake , H. , Inase , M. , & Tanji , J. ( 1991 ). Neuronal activity in the primate premotor, supplementary, and precentral motor cortex during visually guided and internally determined sequential movements . Journal of neurophysiology , 66 ( 3 ), 705 – 718 . OpenUrl CrossRef PubMed Web of Science ↵ Mushiake , H. , Saito , N. , Sakamoto , K. , Itoyama , Y. , & Tanji , J. ( 2006 ). Activity in the lateral prefrontal cortex reflects multiple steps of future events in action plans . Neuron , 50 ( 4 ), 631 – 641 . OpenUrl CrossRef PubMed Web of Science ↵ Ortega , P. A. & Braun , D. A. ( 2013 ). Thermodynamics as a theory of decision-making with information-processing costs . Proceedings of the Royal Society A: Mathematical, Physical and Engineering Sciences , 469 ( 2153 ), 20120683 . OpenUrl ↵ Panichello , M. F. , Jonikaitis , D. , Oh , Y. J. , Zhu , S. , Trepka , E. B. , & Moore , T. ( 2024 ). Intermittent rate coding and cue-specific ensembles support working memory . Nature , 636 ( 8042 ), 422 – 429 . OpenUrl CrossRef PubMed ↵ Parisi , G. ( 1988 ). Statistical field theory. Frontiers in physics . Addison-Wesley . ↵ Parr , T. , Friston , K. , & Pezzulo , G. ( 2024 ). Generative models for sequential dynamics in active inference . Cognitive Neurodynamics , 18 ( 6 ), 3259 – 3272 . OpenUrl PubMed ↵ Parr , T. , Limanowski , J. , Rawji , V. , & Friston , K. ( 2021 ). The computational neurology of movement under active inference . Brain , 144 ( 6 ), 1799 – 1818 . OpenUrl CrossRef PubMed ↵ Parr , T. , Marković , D. , Kiebel , S. J. , & Friston , K. J. ( 2019 ). Neuronal message passing using mean-field, bethe, and marginal approximations . Scientific Reports , 9 ( 1 ), 1109 . OpenUrl PubMed ↵ Parr , T. , Pezzulo , G. , & Friston , K. J. ( 2022 ). Active inference: the free energy principle in mind, brain, and behavior . MIT Press . ↵ Pearl , J. ( 2022 ). Reverend bayes on inference engines: A distributed hierarchical approach . Probabilistic and causal inference: the works of Judea Pearl , 129 – 138 . ↵ Pezzulo , G. , Donnarumma , F. , Maisto , D. , & Stoianov , I. ( 2019 ). Planning at decision time and in the background during spatial navigation . Current opinion in behavioral sciences , 29 , 69 – 76 . OpenUrl ↵ Pezzulo , G. , Kemere , C. , & Van Der Meer , M. A. ( 2017 ). Internally generated hippocampal sequences as a vantage point to probe future-oriented cognition . Annals of the New York Academy of Sciences , 1396 ( 1 ), 144 – 165 . OpenUrl CrossRef PubMed ↵ Pezzulo , G. , Rigoli , F. , & Friston , K. J. ( 2018 ). Hierarchical active inference: a theory of motivated control . Trends in cognitive sciences , 22 ( 4 ), 294 – 306 . OpenUrl CrossRef PubMed ↵ Pezzulo , G. , Van der Meer , M. A. , Lansink , C. S. , & Pennartz , C. M. ( 2014 ). Internally generated sequences in learning and executing goal-directed behavior . Trends in cognitive sciences , 18 ( 12 ), 647 – 657 . OpenUrl CrossRef PubMed Web of Science ↵ Proietti , R. , Parr , T. , Tessari , A. , Friston , K. , & Pezzulo , G. ( 2025 ). Active inference and cognitive control: Balancing deliberation and habits through precision optimization . Physics of Life Reviews . ↵ Proietti , R. , Pezzulo , G. , & Tessari , A. ( 2023 ). An active inference model of hierarchical action understanding, learning and imitation . Physics of Life Reviews , 46 , 92 – 118 . OpenUrl PubMed ↵ Rabinovich , M. I. , Varona , P. , Tristan , I. , & Afraimovich , V. S. ( 2014 ). Chunking dynamics: heteroclinics in mind . Frontiers in computational neuroscience , 8 , 22 . OpenUrl PubMed ↵ Raju , R. V. , Guntupalli , J. S. , Zhou , G. , Wendelken , C. , Lázaro-Gredilla , M. , & George , D. ( 2024 ). Space is a latent sequence: A theory of the hippocampus . Science Advances , 10 ( 31 ), eadm8470 . OpenUrl CrossRef PubMed ↵ Saito , N. , Mushiake , H. , Sakamoto , K. , Itoyama , Y. , & Tanji , J. ( 2005 ). Representation of immediate and final behavioral goals in the monkey prefrontal cortex during an instructed delay period . Cerebral Cortex , 15 ( 10 ), 1535 – 1546 . OpenUrl CrossRef PubMed Web of Science ↵ Schuck , N. W. , Cai , M. B. , Wilson , R. C. , & Niv , Y. ( 2016 ). Human orbitofrontal cortex represents a cognitive map of state space . Neuron , 91 ( 6 ), 1402 – 1412 . OpenUrl CrossRef PubMed ↵ Schwartenbeck , P. , Baram , A. , Liu , Y. , Mark , S. , Muller , T. , Dolan , R. , Botvinick , M. , Kurth-Nelson , Z. , & Behrens , T. ( 2023 ). Generative replay underlies compositional inference in the hippocampal-prefrontal circuit . Cell , 186 ( 22 ), 4885 – 4897 . OpenUrl CrossRef PubMed ↵ Stachenfeld , K. L. , Botvinick , M. M. , & Gershman , S. J. ( 2017 ). The hippocampus as a predictive map . Nature neuroscience , 20 ( 11 ), 1643 – 1653 . OpenUrl CrossRef PubMed ↵ Stoianov , I. , Maisto , D. , & Pezzulo , G. ( 2022 ). The hippocampal formation as a hierarchical generative model supporting generative replay and continual learning . Progress in Neurobiology , 217 , 102329 . OpenUrl CrossRef PubMed ↵ Stoianov , I. P. , Pennartz , C. M. , Lansink , C. S. , & Pezzulo , G. ( 2018 ). Model-based spatial navigation in the hippocampus-ventral striatum circuit: A computational analysis . PLoS computational biology , 14 ( 9 ), e1006316 . OpenUrl PubMed ↵ Sussillo , D. , Churchland , M. M. , Kaufman , M. T. , & Shenoy , K. V. ( 2015 ). A neural network that finds a naturalistic solution for the production of muscle activity . Nature neuroscience , 18 ( 7 ), 1025 – 1033 . OpenUrl CrossRef PubMed ↵ Van de Maele , T. , Dhoedt , B. , Verbelen , T. , & Pezzulo , G. ( 2024 ). A hierarchical active inference model of spatial alternation tasks and the hippocampal-prefrontal circuit . Nature Communications , 15 ( 1 ), 9892 . OpenUrl PubMed ↵ Wang , M. Z. & Hayden , B. Y. ( 2021 ). Latent learning, cognitive maps, and curiosity . Current Opinion in Behavioral Sciences , 38 , 1 – 7 . OpenUrl PubMed ↵ Whittington , J. C. , Dorrell , W. , Behrens , T. E. , Ganguli , S. , & El-Gaby , M. ( 2025 ). A tale of two algorithms: Structured slots explain prefrontal sequence memory and are unified with hippocampal cognitive maps . Neuron , 113 ( 2 ), 321 – 333 . OpenUrl CrossRef PubMed ↵ Wilson , R. C. , Takahashi , Y. K. , Schoenbaum , G. , & Niv , Y. ( 2014 ). Orbitofrontal cortex as a cognitive map of task space . Neuron , 81 ( 2 ), 267 – 279 . OpenUrl CrossRef PubMed Web of Science ↵ Xie , Y. , Hu , P. , Li , J. , Chen , J. , Song , W. , Wang , X.-J. , Yang , T. , Dehaene , S. , Tang , S. , Min , B. , et al. ( 2022 ). Geometry of sequence working memory in macaque prefrontal cortex . Science , 375 ( 6581 ), 632 – 639 . OpenUrl CrossRef PubMed View the discussion thread. Back to top Previous Next Posted November 26, 2025. Download PDF Email Thank you for your interest in spreading the word about bioRxiv. NOTE: Your email address is requested solely to identify you as the sender of this article. Your Email * Your Name * Send To * Enter multiple addresses on separate lines or separate them with commas. You are going to email the following Inferential planning in the frontal cortex Message Subject (Your Name) has forwarded a page to you from bioRxiv Message Body (Your Name) thought you would like to see this page from the bioRxiv website. Your Personal Message CAPTCHA This question is for testing whether or not you are a human visitor and to prevent automated spam submissions. Share Inferential planning in the frontal cortex Francesco Donnarumma , Thomas Parr , Karl Friston , James Whittington , Giovanni Pezzulo bioRxiv 2025.11.26.690672; doi: https://doi.org/10.1101/2025.11.26.690672 Share This Article: Copy Citation Tools Inferential planning in the frontal cortex Francesco Donnarumma , Thomas Parr , Karl Friston , James Whittington , Giovanni Pezzulo bioRxiv 2025.11.26.690672; doi: https://doi.org/10.1101/2025.11.26.690672 Citation Manager Formats BibTeX Bookends EasyBib EndNote (tagged) EndNote 8 (xml) Medlars Mendeley Papers RefWorks Tagged Ref Manager RIS Zotero Tweet Widget Facebook Like Google Plus One Subject Area Neuroscience Subject Areas All Articles Animal Behavior and Cognition (7636) Biochemistry (17704) Bioengineering (13898) Bioinformatics (41967) Biophysics (21460) Cancer Biology (18599) Cell Biology (25525) Clinical Trials (138) Developmental Biology (13384) Ecology (19909) Epidemiology (2067) Evolutionary Biology (24326) Genetics (15613) Genomics (22512) Immunology (17740) Microbiology (40423) Molecular Biology (17191) Neuroscience (88645) Paleontology (667) Pathology (2835) Pharmacology and Toxicology (4825) Physiology (7646) Plant Biology (15158) Scientific Communication and Education (2046) Synthetic Biology (4302) Systems Biology (9825) Zoology (2271)
Text is read by the "Ask this paper" AI Q&A widget below. Extraction quality varies by source — PMC NXML preserves structure cleanly, OA-HTML may include some navigation residue, and OA-PDF can have broken hyphenation. The publisher copy (via DOI) is the canonical version.
My notes (saved in your browser only)
Ask this paper
Answers must be backed by verbatim quotes from this paper's full text. Hallucinated quotes are dropped automatically; if no verbatim passage answers the question, we say so. How this works
Funding
- funders
- [{'doi': None, 'name': 'European Research Council', 'awards': ['820213']}, {'doi': None, 'name': 'European Research Council', 'awards': ['101222868']}, {'doi': None, 'name': None, 'awards': ['226793/Z/22/Z']}, {'doi': None, 'name': 'NIHR Academic Clinical Fellowship', 'awards': ['ACF-2023-13-013']}]
Citation neighborhood (sparse)
Too few in-corpus citations on either side for a chart; here are the lists.
Cites (4)
References (86)
- A Cellular Basis for Mapping Behavioural Structure via crossref
- Active Inference and Cognitive Control: Balancing Deliberation and Habits through Precision Optimization via crossref
- A mechanistic theory of planning in prefrontal cortex via crossref
- A theory of multineuronal dimensionality, dynamics and measurement via crossref
- doi:10.1016/j.neuron.2012.10.038 via crossref
- doi:10.1016/j.neuron.2018.10.002 via crossref
- doi:10.1016/j.tics.2012.08.006 via crossref
- doi:10.1523/jneurosci.2110-07.2007 via crossref
- doi:10.1037/0033-295x.113.2.201 via crossref
- doi:10.1016/j.tics.2004.07.003 via crossref
- doi:10.1016/j.neuron.2024.07.024 via crossref
- doi:10.1016/s1364-6613(03)00197-9 via crossref
- doi:10.1038/s41467-024-52240-6 via crossref
- doi:10.1016/j.plrev.2025.08.008 via crossref
- doi:10.1038/35097575 via crossref
- doi:10.1006/cogp.1996.0008 via crossref
- doi:10.1207/s15516709cog1402_1 via crossref
- doi:10.1038/s41586-025-09226-1 via crossref
- doi:10.1146/annurev-neuro-072116-031538 via crossref
- doi:10.1162/neco_a_00912 via crossref
- doi:10.1016/j.neubiorev.2016.06.022 via crossref
- doi:10.1016/j.biopsycho.2024.108891 via crossref
- doi:10.1162/netn_a_00018 via crossref
- doi:10.1162/necoa01738 via crossref
- doi:10.1073/pnas.0804451105 via crossref
- doi:10.1371/journal.pcbi.1000532 via crossref
- doi:10.1126/sciadv.adr6698 via crossref
- doi:10.1038/s41467-021-22559-5 via crossref
- doi:10.1523/jneurosci.02-11-01527.1982 via crossref
- doi:10.1002/9781119159193.ch33 via crossref
- doi:10.1016/0028-3932(95)90866-p via crossref
- doi:10.1038/s42003-021-02994-2 via crossref
- doi:10.1371/journal.pcbi.1004523 via crossref
- doi:10.1038/s41593-024-01675-7 via crossref
- doi:10.1007/s00422-018-0753-2 via crossref
- doi:10.1007/s10994-012-5278-7 via crossref
- doi:10.1016/j.ceb.2015.08.002 via crossref
- doi:10.1126/science.1088545 via crossref
- doi:10.7554/elife.90874 via crossref
- doi:10.1162/089976602760407955 via crossref
- doi:10.3758/s13423-024-02600-6 via crossref
- doi:10.1038/nature12742 via crossref
- doi:10.1016/j.neuron.2021.12.018 via crossref
- doi:10.1146/annurev.neuro.24.1.167 via crossref
- doi:10.1038/nn.4613 via crossref
- doi:10.1152/jn.1991.66.3.705 via crossref
- doi:10.1016/j.neuron.2006.03.045 via crossref
- doi:10.1098/rspa.2012.0683 via crossref
- doi:10.1038/s41586-024-08139-9 via crossref
- doi:10.1063/1.2811677 via crossref
- doi:10.1007/s11571-023-09963-x via crossref
- doi:10.3389/fncom.2018.00090 via crossref
- doi:10.1162/necoa01102 via crossref
- doi:10.1093/brain/awab085 via crossref
- doi:10.1038/s41598-018-38246-3 via crossref
- doi:10.7551/mitpress/12441.001.0001 via crossref
- doi:10.1016/b978-0-08-051489-5.50008-4 via crossref
- doi:10.1145/3501714.3501727 via crossref
- doi:10.1016/j.cobeha.2019.04.009 via crossref
- doi:10.1111/nyas.13329 via crossref
- doi:10.1016/j.tics.2018.01.009 via crossref
- doi:10.1016/j.tics.2014.06.011 via crossref
- doi:10.1371/journal.pcbi.1013180 via crossref
- doi:10.1016/j.plrev.2023.05.012 via crossref
- doi:10.1126/sciadv.adm8470 via crossref
- doi:10.1016/j.patter.2022.100555 via crossref
- doi:10.1038/nature12160 via crossref
- doi:10.1093/cercor/bhi032 via crossref
- doi:10.1016/j.neuron.2016.08.019 via crossref
- doi:10.1016/j.cell.2023.09.004 via crossref
- doi:10.1162/neco_a_01108 via crossref
- doi:10.1038/nn.4650 via crossref
- doi:10.1016/j.pneurobio.2022.102329 via crossref
- doi:10.1371/journal.pcbi.1006316 via crossref
- doi:10.1038/nn.4042 via crossref
- doi:10.1126/science.adp6091 via crossref
- doi:10.1038/s41467-024-54257-3 via crossref
- doi:10.1016/j.cobeha.2020.06.003 via crossref
- doi:10.1016/j.neuron.2024.10.017 via crossref
- doi:10.1016/j.neuron.2013.11.005 via crossref
- doi:10.1126/science.abm0204 via crossref
- doi:10.1073/pnas.162485599 via crossref
- doi:10.1109/tit.2005.850085 via crossref
- doi:10.1016/j.neuron.2016.03.037. via crossref
- doi:10.1016/j.neuron.2020.11.024 via crossref
- doi:10.1162/jocn_a_01662 via crossref
Source provenance
- crossref
- last seen: 2026-07-20T07:00:56.711288+00:00
- europepmc
- last seen: 2026-05-20T01:45:00.602351+00:00
- unpaywall
- last seen: 2026-05-21T02:00:01.467718+00:00
License: CC-BY-4.0