<?xml version="1.0" encoding="utf-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="2.3" xml:lang="EN">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Behav. Neurosci.</journal-id>
<journal-title>Frontiers in Behavioral Neuroscience</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Behav. Neurosci.</abbrev-journal-title>
<issn pub-type="epub">1662-5153</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fnbeh.2024.1399394</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Behavioral Neuroscience</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Cognitive mechanisms of learning in sequential decision-making under uncertainty: an experimental and theoretical approach</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<name><surname>Cecchini</surname> <given-names>Gloria</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="corresp" rid="c001"><sup>&#x002A;</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/2671447/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/conceptualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/formal-analysis/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/software/"/>
<role content-type="https://credit.niso.org/contributor-roles/validation/"/>
<role content-type="https://credit.niso.org/contributor-roles/visualization/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>DePass</surname> <given-names>Michael</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/2773458/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/data-curation/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Baspinar</surname> <given-names>Emre</given-names></name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/1885832/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Andujar</surname> <given-names>Marta</given-names></name>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/2721703/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Ramawat</surname> <given-names>Surabhi</given-names></name>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/2188087/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Pani</surname> <given-names>Pierpaolo</given-names></name>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/99336/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Ferraina</surname> <given-names>Stefano</given-names></name>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/1534/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Destexhe</surname> <given-names>Alain</given-names></name>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/10022/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Moreno-Bote</surname> <given-names>Rub&#x00E9;n</given-names></name>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="aff" rid="aff5"><sup>5</sup></xref>
<uri xlink:href="https://loop.frontiersin.org/people/66667/overview"/>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Cos</surname> <given-names>Ignasi</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff5"><sup>5</sup></xref>
<role content-type="https://credit.niso.org/contributor-roles/funding-acquisition/"/>
<role content-type="https://credit.niso.org/contributor-roles/investigation/"/>
<role content-type="https://credit.niso.org/contributor-roles/methodology/"/>
<role content-type="https://credit.niso.org/contributor-roles/supervision/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-original-draft/"/>
<role content-type="https://credit.niso.org/contributor-roles/writing-review-editing/"/>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>Facultat de Matem&#x00E0;tiques i Inform&#x00E0;tica, Universitat de Barcelona</institution>, <addr-line>Barcelona</addr-line>, <country>Spain</country></aff>
<aff id="aff2"><sup>2</sup><institution>Center for Brain and Cognition, DTIC, Universitat Pompeu Fabra</institution>, <addr-line>Barcelona</addr-line>, <country>Spain</country></aff>
<aff id="aff3"><sup>3</sup><institution>CNRS, Institute of Neuroscience (NeuroPSI), Paris-Saclay University</institution>, <addr-line>Saclay</addr-line>, <country>France</country></aff>
<aff id="aff4"><sup>4</sup><institution>Department of Physiology and Pharmacology, Sapienza University of Rome</institution>, <addr-line>Rome</addr-line>, <country>Italy</country></aff>
<aff id="aff5"><sup>5</sup><institution>Serra-Hunter Fellow Programme</institution>, <addr-line>Barcelona</addr-line>, <country>Spain</country></aff>
<author-notes>
<fn fn-type="edited-by" id="fn0001">
<p>Edited by: Mario Trevi&#x00F1;o, University of Guadalajara, Mexico</p>
</fn>
<fn fn-type="edited-by" id="fn0002">
<p>Reviewed by: Jeffrey C. Erlich, University College London, United Kingdom</p>
<p>Tanya Gupta, University of California, Los Angeles, United States</p>
</fn>
<corresp id="c001">&#x002A;Correspondence: Gloria Cecchini, <email>gloria.cecchini@ub.edu</email></corresp>
</author-notes>
<pub-date pub-type="epub">
<day>12</day>
<month>08</month>
<year>2024</year>
</pub-date>
<pub-date pub-type="collection">
<year>2024</year>
</pub-date>
<volume>18</volume>
<elocation-id>1399394</elocation-id>
<history>
<date date-type="received">
<day>11</day>
<month>03</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>19</day>
<month>07</month>
<year>2024</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x00A9; 2024 Cecchini, DePass, Baspinar, Andujar, Ramawat, Pani, Ferraina, Destexhe, Moreno-Bote and Cos.</copyright-statement>
<copyright-year>2024</copyright-year>
<copyright-holder>Cecchini, DePass, Baspinar, Andujar, Ramawat, Pani, Ferraina, Destexhe, Moreno-Bote and Cos</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/">
<p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) and the copyright owner(s) are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p>
</license>
</permissions>
<abstract>
<p>Learning to make adaptive decisions involves making choices, assessing their consequence, and leveraging this assessment to attain higher rewarding states. Despite vast literature on value-based decision-making, relatively little is known about the cognitive processes underlying decisions in highly uncertain contexts. Real world decisions are rarely accompanied by immediate feedback, explicit rewards, or complete knowledge of the environment. Being able to make informed decisions in such contexts requires significant knowledge about the environment, which can only be gained via exploration. Here we aim at understanding and formalizing the brain mechanisms underlying these processes. To this end, we first designed and performed an experimental task. Human participants had to learn to maximize reward while making sequences of decisions with only basic knowledge of the environment, and in the absence of explicit performance cues. Participants had to rely on their own internal assessment of performance to reveal a covert relationship between their choices and their subsequent consequences to find a strategy leading to the highest cumulative reward. Our results show that the participants&#x2019; reaction times were longer whenever the decision involved a future consequence, suggesting greater introspection whenever a delayed value had to be considered. The learning time varied significantly across participants. Second, we formalized the neurocognitive processes underlying decision-making within this task, combining mean-field representations of competing neural populations with a reinforcement learning mechanism. This model provided a plausible characterization of the brain dynamics underlying these processes, and reproduced each aspect of the participants&#x2019; behavior, from their reaction times and choices to their learning rates. In summary, both the experimental results and the model provide a principled explanation to how delayed value may be computed and incorporated into the neural dynamics of decision-making, and to how learning occurs in these uncertain scenarios.</p>
</abstract>
<kwd-group>
<kwd>decision-making</kwd>
<kwd>learning</kwd>
<kwd>cognition</kwd>
<kwd>computational modeling</kwd>
<kwd>consequence</kwd>
<kwd>uncertanty</kwd>
<kwd>neural dynamics</kwd>
<kwd>behavior</kwd>
</kwd-group>
<counts>
<fig-count count="10"/>
<table-count count="2"/>
<equation-count count="5"/>
<ref-count count="98"/>
<page-count count="22"/>
<word-count count="19758"/>
</counts>
<custom-meta-wrap>
<custom-meta>
<meta-name>section-at-acceptance</meta-name>
<meta-value>Learning and Memory</meta-value>
</custom-meta>
</custom-meta-wrap>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="sec1">
<label>1</label>
<title>Introduction</title>
<p>The brain mechanisms involved in decision-making have been extensively studied in the last decades [reviewed in (<xref ref-type="bibr" rid="ref28">Gold and Shadlen, 2007</xref>; <xref ref-type="bibr" rid="ref89">Wang, 2008</xref>)]. Many studies focused on characterizing the neural dynamics of reward processing (<xref ref-type="bibr" rid="ref61">Padoa-Schioppa, 2011</xref>; <xref ref-type="bibr" rid="ref87">Wallis and Kennerley, 2011</xref>; <xref ref-type="bibr" rid="ref27">Gluth et al., 2014</xref>), visual discrimination (<xref ref-type="bibr" rid="ref71">Shadlen and Newsome, 1996</xref>; <xref ref-type="bibr" rid="ref72">Shadlen and Newsome, 2001</xref>; <xref ref-type="bibr" rid="ref66">Roitman and Shadlen, 2002</xref>), and other aspects of option assessment during value-based decision-making (<xref ref-type="bibr" rid="ref63">Pastor-Bernier and Cisek, 2011</xref>; <xref ref-type="bibr" rid="ref86">Wallis, 2011</xref>; <xref ref-type="bibr" rid="ref11">Cai and Padoa-Schioppa, 2019</xref>; <xref ref-type="bibr" rid="ref12">Carroll et al., 2019</xref>). Other tasks were developed to study decisions in the context of short-term memory (<xref ref-type="bibr" rid="ref74">Siegel et al., 2009</xref>), and cost-risk trade-off (<xref ref-type="bibr" rid="ref41">Kahneman and Tversky, 1979</xref>; <xref ref-type="bibr" rid="ref6">Birnbaum, 2008</xref>; <xref ref-type="bibr" rid="ref23">Eichberger and Pasichnichenko, 2021</xref>). In most of these contexts, choice outcomes are immediately experienced. This feature makes calculating costs and benefits straightforward, as all the necessary information is directly and immediately available to the decision maker for calculation (<xref ref-type="bibr" rid="ref48">Kurniawan et al., 2013</xref>; <xref ref-type="bibr" rid="ref75">Skvortsova et al., 2014</xref>; <xref ref-type="bibr" rid="ref3">Apps et al., 2015</xref>; <xref ref-type="bibr" rid="ref83">Thura and Cisek, 2016</xref>). However, a complete account of value-based choice behavior requires understanding the brain mechanisms underlying the detection and computation of non-immediate consequences of choices, and of the use of this information to guide subsequent decision strategies. Despite the rich literature in cognitive decision-making and the fact that long-term consequence is a critical concern in our daily decision-making processes, the dynamics of its operation are not fully understood, and have not been incorporated into state-of-the-art models of decision-making (<xref ref-type="bibr" rid="ref10">Brunel and Wang, 2001</xref>; <xref ref-type="bibr" rid="ref96">Wong and Wang, 2006</xref>; <xref ref-type="bibr" rid="ref95">Wong et al., 2007</xref>). Most previous models work only for independent trials by considering value and/or accumulation of evidence about choice alternatives (<xref ref-type="bibr" rid="ref21">Drugowitsch et al., 2012</xref>). They often do not, however, take into consideration the memory of recent past or the long-term effects of decisions in the context of brain dynamics. By contrast, studies on hierarchical decision-making show that when choices are repeatedly made along nodes of the same decision-tree, they tend to integrate elements of subsequent nodes (<xref ref-type="bibr" rid="ref39">Hyafil and Moreno-Bote, 2017</xref>). In other words, the assessment of options during decisions incorporates elements of subsequent branching points. However, for these decisions to be informed, exploration and ultimately knowledge about future nodes is required.</p>
<p>Here we are interested in formalizing the brain mechanisms underlying how this exploration leads to information gain when the strategy is non-obvious. In other words, which are the brain operations involved in considering the consequence of choices during sequences of decisions. In this scenario, the case when the immediate most rewarding choice leads to lower long-term reward is of particular interest, as participants must anticipate that the cost of choosing lower value options results in increased delayed reward and higher cumulative reward overall. Moreover, if this relationship is covert, what are the cognitive mechanisms that enable us to learn the optimal strategy? Furthermore, how does the learning occur in the absence of explicit performance feedback?</p>
<p>To answer these questions, we developed the <italic>consequential task</italic>. Consecutive perceptual decision-making trials were organized into groups of dependent trials, where the choice made in one trial had a consequence on the next by determining the available choice options. How does the complexity of a perceptual decision-making task augment when combined with consequence assessment? First, consequence-based decisions (i.e., decisions in which optimal performance can only be achieved after acquiring knowledge of future nodes) require an increased temporal span of consideration, and, consequently, involve a more uncertain and broader set of factors to examine. This typically results in more computationally demanding option evaluation (<xref ref-type="bibr" rid="ref84">Trommersh&#x00E4;user et al., 2008</xref>; <xref ref-type="bibr" rid="ref58">Nagengast et al., 2011</xref>; <xref ref-type="bibr" rid="ref60">O&#x2019;Brien and Ahmed, 2015</xref>; <xref ref-type="bibr" rid="ref44">Kirchler et al., 2017</xref>), longer deliberation, and often poorer decision accuracy (<xref ref-type="bibr" rid="ref70">Schuck-Paim and Kacelnik, 2007</xref>; <xref ref-type="bibr" rid="ref22">Drugowitsch et al., 2016</xref>). Second, making decisions based on gauging choice consequence involves a range of cognitive processes broader than those involved in immediate sensory-motor decisions (<xref ref-type="bibr" rid="ref15">Cisek et al., 2009</xref>; <xref ref-type="bibr" rid="ref20">Donner et al., 2009</xref>), with particular emphasis on value integration (<xref ref-type="bibr" rid="ref14">Cisek and Kalaska, 2005</xref>; <xref ref-type="bibr" rid="ref62">Park et al., 2011</xref>), metacognitive processing (<xref ref-type="bibr" rid="ref45">Klaes et al., 2011</xref>; <xref ref-type="bibr" rid="ref29">Goodwin et al., 2012</xref>) and long-term working memory (<xref ref-type="bibr" rid="ref13">Cavanagh et al., 2018</xref>; <xref ref-type="bibr" rid="ref5">Barbosa et al., 2020</xref>). Though long-term consequence assessment may be viewed as a time extended version of immediate action outcome evaluation, significant doubts remain regarding the core cognitive and neural processes underlying this ability (<xref ref-type="bibr" rid="ref4">Balasubramani and Hayden, 2018</xref>).</p>
<p>To investigate the cognitive processes underlying consequence-based decision-making, we carried out a combined experimental and theoretical study. In the first part of this work, we designed a decision-making task, the <italic>consequential task</italic>, to characterize consequence-based option assessment. In brief, in the consequential task, consequence takes the form of increases/decreases in future reward value options as a function of participants&#x2019; choices. The nature of this inter-trial dependence was not disclosed in the instructions given to the participants, and no explicit performance feedback was provided. The absence of explicit learning cues was intended to force the participants to rely on their own subjective performance feedback to infer the delayed consequence of their decisions.</p>
<p>In the second part of our study, we provided a theoretical framework of the cognitive and neural processes required for consequence-based decision-making, including the patterns of inhibition and of far-sighted consequence assessment required to acquire the most reward across trials. The model was organized in three layers. The bottom layer, in line with the Amari, Wilson-Cowan and Wong-Wang models (<xref ref-type="bibr" rid="ref94">Wilson and Cowan, 1972</xref>; <xref ref-type="bibr" rid="ref78">Soltani et al., 2006</xref>; <xref ref-type="bibr" rid="ref96">Wong and Wang, 2006</xref>; <xref ref-type="bibr" rid="ref90">Webb et al., 2011</xref>; <xref ref-type="bibr" rid="ref54">Marcos et al., 2013</xref>; <xref ref-type="bibr" rid="ref34">Hert&#x00E4;g et al., 2014</xref>), described the neural dynamics of binary decision-making by means of two populations of neurons. The middle and top layers modeled an oversight mechanism for the assessment of consequence across groups of trials and the learning mechanism as a function of reward value across trials. This model reproduced the full range of behavioral observations across the different participants accurately while predicting a plausible neural implementation of the processes underlying the learning of consequence-based decision-making. In particular, our model described how the metacognitive assessment of consequence extends from short to long-term value prediction through an oversight mechanism that monitors predicted performance.</p>
</sec>
<sec sec-type="results" id="sec2">
<label>2</label>
<title>Results</title>
<sec id="sec3">
<label>2.1</label>
<title>Task design</title>
<p>In this section we describe the consequential task and, more specifically, how it is designed to tap into the cognitive mechanisms involved in learning delayed consequences in the absence of explicit performance feedback. In this task, 28 healthy participants were instructed to choose one of two stimuli presented left and right on a screen. The stimuli represented partially filled containers of water and reward value was directly proportional to the amount of water in each container. The participants reported their choices by moving the computer mouse&#x2019;s cursor from the central cue to the chosen stimulus (see <xref ref-type="fig" rid="fig1">Figure 1</xref> and Materials and Methods for a thorough description). Participants were only paid a show-up fee and were, thus, not monetarily incentivized to perform well.</p>
<fig position="float" id="fig1">
<label>Figure 1</label>
<caption>
<p>Time-course of a typical horizon 1 episode of the consequential decision-making task. <bold>(A)</bold> The episode consists of two dependent trials. The first starts with the message &#x201C;New Episode Starting&#x201D; in the center-top of the screen, a circle surrounding a cross in the center (central target), and a half full progress bar at the bottom of the screen. The progress bar indicates the current trial within the episode (for horizon 1, 50% during the first trial, 100% during the second trial). After holding for 500&#x2009;ms, the left or right (chosen at random) stimulus is shown, followed by its complementary stimulus 500&#x2009;ms later. Both stimuli are shown simultaneously 500&#x2009;ms later which serves as the GO signal. At GO, the participant has to slide the mouse from the central target to the bar of their choosing. Once the selected target is reached, a yellow dot appears over that target. The second trial follows the same pattern as the first. See Methods for more details. <bold>(B)</bold> Construction scheme for the size of the stimuli in each episode. The first trial within the episode consists of 2 stimuli of size M&#x2009;+&#x2009;d/2 and M&#x2212;d/2. The second trial within the episode depends on the selection made in the previous trial. If the first selected stimulus is M&#x2212;d/2 (following symbol &#x201C;-&#x201D; in the figure), then the second trial consists of stimuli with size M&#x2009;+&#x2009;G&#x2009;+&#x2009;d/2 and M&#x2009;+&#x2009;G&#x2212;d/2, otherwise M-G&#x2009;+&#x2009;d/2 and M-G-d/2 (following symbol &#x201C;+&#x201D; in the figure). The cumulative reward value of the episode can therefore assume 4 distinct values (ordered from best to worst): 2&#x2009;M&#x2009;+&#x2009;G, 2&#x2009;M&#x2009;+&#x2009;G-d, 2&#x2009;M-G&#x2009;+&#x2009;d, and 2&#x2009;M-G. See Methods for more details on the values of M, G, d.</p>
</caption>
<graphic xlink:href="fnbeh-18-1399394-g001.tif"/>
</fig>
<p>Since consequence depends on a predictive assessment of future contexts, the task was organized into two block types. In the first, trials required one-shot decisions, purely independent of one another. Similar to most decision-making tasks, the reward value in this block type could be maximized by picking the stimulus associated with the most reward value in each trial, i.e., choosing the larger of the two blue bars. However, in the second block type, trials were grouped into pairs or triads of dependent trials. We called each group of consecutive trials an episode to signify the boundary of dependence between them, and defined the notion of horizon (<italic>n<sub>H</sub></italic>) as a metric for the depth of consequence to be expected for that episode. In other words, <italic>n<sub>H</sub></italic> equaled the number of dependent trials following the first trial of the episode. For example, for <italic>n<sub>H</sub></italic>&#x2009;=&#x2009;1 an episode consisted of 2 trials, with the second depending on the first. The nature of the dependence between trials of an episode was such that the mean reward values of the stimuli in the second/third trial were systematically increased or decreased based on the participant&#x2019;s choice in the preceding trial. Choosing the larger stimulus value led to a reduction of stimuli values in the subsequent trial whereas choosing the smaller stimulus in the first trial led to an increase (<xref ref-type="fig" rid="fig1">Figure 1B</xref>). The increment/reduction amount (<italic>G</italic>) was a constant and chosen such that selecting the larger stimulus in the first trial could never compensate for the loss in future reward value. In other words, acquiring the maximum cumulative reward value in each episode required choosing &#x201C;big&#x201D; in single trial episodes (horizon <italic>n<sub>H</sub></italic>&#x2009;=&#x2009;0), and choosing &#x201C;small&#x201D; in all trials of <italic>n<sub>H</sub></italic>&#x2009;=&#x2009;1 and <italic>n<sub>H</sub></italic>&#x2009;=&#x2009;2 episodes except the last, in which &#x201C;big&#x201D; should be chosen.</p>
<p>The consequential task design enables investigation into the role of perceived consequence during sequential decision-making. Consequence, in this context, refers to the influence of a choice on the stimuli values in the trial next. The post-decision stimuli heights function as a form of feedback which participants must learn to interpret in order to become aware of and evaluate the consequences associated with particular choices. Performance feedback, however, is absent from the task in that participants are never presented with cues indicating whether they are behaving optimally. This absence required participants to evaluate their own performance based on their experience during task execution. Importantly, participants were not informed of the nature of the inter-trial dependence and had to discover it on their own via exploration. Explicit performance feedback might have had the undesirable effect of participants focusing on finding the specific sequence of choices yielding optimal performance feedback, without having to learn the dependence between their decisions and the subsequent trials. In other words, an explicit measure of performance might have reduced the task to an explicit trial-and-error test in which participants would experiment with different sequences of choices (&#x201C;big-small,&#x201D; &#x201C;small-big,&#x201D; etc.) until finding the sequence leading to maximum performance, rather than learning to evaluate each option&#x2019;s consequence in terms of their prediction of future reward. In contrast, the absence of performance feedback obligated participants to create an internal sense of assessment, which could only rely on two mechanisms: the sensory perception of the systematic stimuli changes in the subsequent trial after each choice, and the exploration of option choices at each trial during the earlier part of each block. The resulting task essentially becomes a measure of learning about delayed consequences associated with each option in the absence of explicit performance feedback.</p>
<p>In summary, for the participants to be able to perform the task, they were informed of the episode-based organization of trials at each block, i.e., the horizon. The instruction to the participant was to find the strategy leading to the most cumulative reward value for each episode and to actively explore their choices. Learning the optimal policy was challenging due to several factors. First, perceptual discrimination was difficult in some trials since the height difference between stimuli could be as low as 1% the height of the container. Second, although participants were informed that their choices may affect future trials within the episode, the nature of this dependency was not specified. This means that from the perspective of the participants, the value of the stimuli offers might at first appear random. Third, explicit performance feedback was omitted from the task after each episode, requiring participants to discover the nature of the inter-trial dependencies via exploration. Further details are shown in the Methods section, and in <xref ref-type="fig" rid="fig1">Figure 1</xref>.</p>
</sec>
<sec id="sec4">
<label>2.2</label>
<title>Behavioral results</title>
<p>Several metrics were extracted from the participants&#x2019; behavioral data: performance (PF), reported choices (CH), reaction time (RT), and visual discrimination (VD) sensitivity. PF was extracted from each episode and assumed values between 0 (worst) and 1 (best). PF was calculated as the percentage of the maximum possible reward value acquired in each episode and is normalized such that PF&#x2009;=&#x2009;0 in episodes wherein the participant acquired the minimum possible reward value. CH was the choice made by the participant in each trial and could take one of two values: small (i.e., smaller stimulus), or large (i.e., larger stimulus). RT was calculated as the time difference between the simultaneous presentation of both stimuli (the GO signal), and the onset of the movement. VD is a measure of each participants&#x2019; ability to visually discriminate between stimuli, i.e., identifying which stimulus is bigger/smaller (see Methods for further details). As shown below, when the difference between stimuli (&#x0394;S) is the smallest, participants were not able to accurately distinguish between stimuli. The &#x0394;S varies between 1 and 20% of the size of the container. Note that for horizon <italic>n<sub>H</sub> =&#x2009;0,</italic> a trial with &#x0394;S&#x2009;=&#x2009;0.01 is perceptually difficult, but if chosen wrong, the difference in the final reward would be small (1%). However, for horizon <italic>n<sub>H</sub> =&#x2009;1 or 2</italic>, choosing the wrong stimulus due to perceptual discrimination has a large impact on the final performance, since it leads to a decrease of the available stimuli in the next trials.</p>
<p>The absence of explicit performance-related feedback at the end of each episode made the task more difficult, and, consequently, not all participants were able to find the optimal strategy. For horizon <italic>n<sub>H</sub> =&#x2009;0,</italic> 26 of the 28 participants learned and applied the optimal strategy, i.e., repeatedly selecting the larger stimulus. In contrast, only 22 participants learned the optimal strategy during horizon <italic>n<sub>H</sub> = 1, 2</italic> blocks, i.e., selecting the larger stimulus in the last trial only.</p>
<p>We analyzed the exploratory strategies the participants employed. In particular, we tested whether participants only considered the size of the stimuli (small/big), or if they also tested other hypotheses involving the order of presentation of the stimuli (first/s) or the location (left/right) of the stimuli. The result of this analysis can be found in the <xref rid="SM1" ref-type="supplementary-material">Supplementary Figure S1</xref>. In brief, participants&#x2019; choices overwhelmingly depended on stimuli size and there was little evidence other factors such as order of presentation or location were seriously considered in the decision-making process. Most participants who did not learn the optimal strategy for <italic>n<sub>H</sub> =&#x2009;1,2</italic> repeatedly chose the larger stimulus for all trials.</p>
<p>In Materials and Methods (subsection Consequential Decision-Making task), we described how the task was structured, and we mentioned that we randomized the order in which participants performed the horizons. This means that, for example, some participants performed <italic>n<sub>H</sub></italic>&#x2009;=&#x2009;2 before <italic>n<sub>H</sub></italic>&#x2009;=&#x2009;0. We wondered if the order of execution of the horizons had an influence on learning. To address this, we performed an analysis comparing learning times for different orders of horizon presentation. The results of this investigation can be found in the <xref rid="SM1" ref-type="supplementary-material">Supplementary Figures S2, S3</xref>. In brief, we discovered that once the optimal strategy was understood in <italic>n<sub>H</sub></italic>&#x2009;=&#x2009;1 or 2, participants generalized the rule and, by abstraction, applied it to the horizon performed afterwards. For this reason, we defined a single learning time per session. We defined <italic>learning time (t<sub>L</sub>)</italic> as the number of episodes that occurred before the optimal strategy was assimilated. We considered the optimal strategy to be assimilated if the participant employed it in at least 9 out of the following 10 episodes, and 75% of the remaining episodes until the end of the block. To account for perceptual discrimination errors (during low VD), we excluded the most difficult episodes in terms of &#x0394;S to calculate the learning time.</p>
<p><xref ref-type="fig" rid="fig2">Figure 2</xref> shows the summary results for all 28 participants. In Panel (a), we show the histogram of their learning times in terms of episodes (<italic>E</italic>). The last histogram bar in <xref ref-type="fig" rid="fig2">Figure 2A</xref> (shown as NL &#x2013; No Learning) represents the 6 participants who never learned the optimal strategy. We divided participants into 4 groups as a function of their learning speed: slow, medium, fast, and those who never learned the optimal strategy.</p>
<fig position="float" id="fig2">
<label>Figure 2</label>
<caption>
<p>Summary behavioral results across participants. <bold>(A)</bold> Histogram of learning times. Learning time is defined as the number of episodes <bold>(E)</bold> throughout the whole session before the optimal strategy was applied repeatedly (see Methods). We identified four groups of participants: fast, medium and slow learners, and participants who did not discover the optimal strategy (NL &#x2013; No Learning). <bold>(B)</bold> Histogram of visual discrimination (VD) calculated by computing the percentage of correct selections of the last 80 episodes, in the horizon 0 block, for only the most difficult trials (&#x0394;<italic>S</italic>&#x2009;=&#x2009;0.01). <bold>(C)</bold> Performance as a function of &#x0394;S, for the trials after the optimal strategy was applied. <bold>(D)</bold> Reaction Time (RT) versus &#x0394;<italic>S</italic>. The more similar the stimuli, the longer participants needed to make a decision. <bold>(E,F)</bold> Regression coefficients for the generalized linear mixed-effects models <inline-formula>
<mml:math id="M1">
<mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi>o</mml:mi>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>~</mml:mo>
<mml:mi>L</mml:mi>
<mml:mo>&#x00B7;</mml:mo>
<mml:mi>E</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>L</mml:mi>
<mml:mo>&#x00B7;</mml:mo>
<mml:mi>&#x0394;</mml:mi>
<mml:mi>S</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>L</mml:mi>
<mml:mo>&#x00B7;</mml:mo>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mi>H</mml:mi>
</mml:msub>
<mml:mo>&#x00B7;</mml:mo>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>E</mml:mi>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:mi mathvariant="normal">(</mml:mi>
<mml:mi>E</mml:mi>
<mml:mi mathvariant="normal">|</mml:mi>
<mml:mi>p</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math id="M2">
<mml:mrow>
<mml:mi>R</mml:mi>
<mml:mi>T</mml:mi>
<mml:mo>~</mml:mo>
<mml:mi>L</mml:mi>
<mml:mo>&#x00B7;</mml:mo>
<mml:mi>E</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>L</mml:mi>
<mml:mo>&#x00B7;</mml:mo>
<mml:mi>&#x0394;</mml:mi>
<mml:mi>S</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>L</mml:mi>
<mml:mo>&#x00B7;</mml:mo>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mi>H</mml:mi>
</mml:msub>
<mml:mo>&#x00B7;</mml:mo>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>E</mml:mi>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:mi mathvariant="normal">(</mml:mi>
<mml:mi>E</mml:mi>
<mml:mi mathvariant="normal">|</mml:mi>
<mml:mi>p</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>, where <inline-formula>
<mml:math id="M3">
<mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi>o</mml:mi>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the probability of making the optimal choice, RT is the reaction time, E is the episode number, <inline-formula>
<mml:math id="M4">
<mml:mrow>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mi>H</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the horizon number, <inline-formula>
<mml:math id="M5">
<mml:mrow>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>E</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the trial within episode, L identifies the group of participants that learned the optimal strategy, <inline-formula>
<mml:math id="M6">
<mml:mrow>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mi>H</mml:mi>
</mml:msub>
<mml:mo>:</mml:mo>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>E</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> is the interaction term, and p is the participant. We used maximum likelihood to estimate the model parameters. Participants were divided into two groups: those who learned the optimal strategy (blue) and those who did not (red), see Panel (a). The statistical difference between learning groups in reported next to the legend.</p>
</caption>
<graphic xlink:href="fnbeh-18-1399394-g002.tif"/>
</fig>
<p><xref ref-type="fig" rid="fig2">Figure 2B</xref> shows the VD, for all difficult trials (smallest &#x0394;S) and participants, where VD was calculated as the percentage of correct choices over the last 80 episodes in the horizon <italic>n<sub>H</sub> =&#x2009;0</italic> block. On average, stimuli were discriminated correctly in 71% of the most difficult trials. This indicates that most participants continued making errors after learning the optimal strategy due to low VD. This is reported in <xref ref-type="fig" rid="fig2">Figure 2C</xref> which shows the grand average and standard error of the PF across subjects as a function of the difficulty level for all episodes following each participant&#x2019;s learning time (<italic>p</italic>&#x2009;=&#x2009;10<sup>&#x2212;12</sup>, F-stat&#x2009;=&#x2009;59). Note that, in <xref ref-type="fig" rid="fig2">Figure 2D</xref>, RT increased as a function of VD (<italic>p</italic>&#x2009;=&#x2009;10<sup>&#x2212;25</sup>, F-stat&#x2009;=&#x2009;160).</p>
<p>The dependence of PF and RT on VD together with the other variables had to be established statistically. To assess the learning process, we quantified the relationship of PF and RT with horizon <italic>n<sub>H</sub></italic>, trial within episode <italic>T<sub>E</sub></italic>, and episode <italic>E</italic>. To obtain consistent results, we adjusted these variables as follows. The trial within episode was reversed chronologically, because the optimal choice for the last <italic>T<sub>E</sub></italic> (large) is the same regardless of the horizon number. Furthermore, regarding the model of PF, we made a per trial adaptation of PF (PF was originally calculated per episode), i.e., the probability of choosing the optimal choice <inline-formula>
<mml:math id="M7">
<mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi>o</mml:mi>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. Finally, to assess differences between learning groups, we introduced the categorical variable L that identified the group of participants that learned the optimal strategy and the ones who did not (as seen in <xref ref-type="fig" rid="fig2">Figure 2A</xref>). We then used a generalized linear mixed effects model (<xref ref-type="bibr" rid="ref85">Verbeke and Molenberghs, 2009</xref>; <xref ref-type="bibr" rid="ref26">Ga&#x0142;ecki and Burzykowski, 2013</xref>) to predict PF and RT. The independent variables for the fixed effects are horizon <italic>n<sub>H</sub></italic>, trial within episode <inline-formula>
<mml:math id="M8">
<mml:mrow>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>E</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, the passage of time expressed in terms of episodes <italic>E,</italic> and &#x0394;S. We set the random effects for the intercept and the episodes grouped by participant <italic>p</italic>; we write the random effects as <inline-formula>
<mml:math id="M9">
<mml:mrow>
<mml:mi mathvariant="normal">(</mml:mi>
<mml:mi>E</mml:mi>
<mml:mi mathvariant="normal">|</mml:mi>
<mml:mi>p</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>. The resulting models are: <inline-formula>
<mml:math id="M10">
<mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi>o</mml:mi>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>~</mml:mo>
<mml:mi>L</mml:mi>
<mml:mo>&#x00B7;</mml:mo>
<mml:mi>E</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>L</mml:mi>
<mml:mo>&#x00B7;</mml:mo>
<mml:mi>&#x0394;</mml:mi>
<mml:mi>S</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>L</mml:mi>
<mml:mo>&#x00B7;</mml:mo>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mi>H</mml:mi>
</mml:msub>
<mml:mo>&#x00B7;</mml:mo>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>E</mml:mi>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:mi mathvariant="normal">(</mml:mi>
<mml:mi>E</mml:mi>
<mml:mi mathvariant="normal">|</mml:mi>
<mml:mi>p</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math id="M11">
<mml:mrow>
<mml:mi>R</mml:mi>
<mml:mi>T</mml:mi>
<mml:mo>~</mml:mo>
<mml:mi>L</mml:mi>
<mml:mo>&#x00B7;</mml:mo>
<mml:mi>E</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>L</mml:mi>
<mml:mo>&#x00B7;</mml:mo>
<mml:mi>&#x0394;</mml:mi>
<mml:mi>S</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>L</mml:mi>
<mml:mo>&#x00B7;</mml:mo>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mi>H</mml:mi>
</mml:msub>
<mml:mo>&#x00B7;</mml:mo>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>E</mml:mi>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:mi mathvariant="normal">(</mml:mi>
<mml:mi>E</mml:mi>
<mml:mi mathvariant="normal">|</mml:mi>
<mml:mi>p</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
<mml:mo>.</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> The regression coefficients, along with their respective group significance, are shown in <xref ref-type="fig" rid="fig2">Figures 2E</xref>,<xref ref-type="fig" rid="fig2">F</xref>. The detailed results of the statistical analysis are reported in Section 5.5. In panel (e), <inline-formula>
<mml:math id="M12">
<mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi>o</mml:mi>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> increases with <inline-formula>
<mml:math id="M13">
<mml:mrow>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>E</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, suggesting that the first trial(s) within the episode are less likely to be guessed right, i.e., choosing the smaller stimuli. This makes sense, since only the early trials within episode required inhibition. Moreover, looking at the amplitude of the regression coefficients, we can see that this effect is even stronger in the no-learning case. The same argument can be made for the dependence with <italic>n<sub>H</sub></italic>. A strong difference between learning and no-learning can be appreciated when considering the time dependence: for the learners group <inline-formula>
<mml:math id="M14">
<mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi>o</mml:mi>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> increases as time goes by, i.e., <italic>E</italic> increases, while it is not significant for the group that did not learn the optimal strategy. The two learning groups are globally statistically different (<italic>p&#x2009;=&#x2009;10<sup>&#x2212;7</sup></italic>). In panel (f), RT shows converse effect directions between learning and no-learning groups for both dependencies on <inline-formula>
<mml:math id="M15">
<mml:mrow>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>E</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and <italic>n<sub>H</sub></italic>. The participants who learned the optimal strategy exhibited longer RT for the earlier trials within the episode, consistent with the need to inhibit the selection of the larger stimulus. Also, the larger the horizon, the longer the RT, opposite to the no-learning group. As expected, RT increases with decreasing &#x0394;S for both groups. The two learning groups are globally statistically different (<italic>p&#x2009;=&#x2009;10<sup>&#x2212;17</sup></italic>).</p>
<p><xref ref-type="fig" rid="fig3">Figure 3</xref> depicts the data from 3 sample participants. In particular we show their PFs, CHs, and RTs metrics, and the order of execution of the different blocks and horizons. Each column corresponds to a participant and each row to a different horizon level. Note that all three participants performed the <italic>n<sub>H</sub> =&#x2009;0</italic> task correctly (<xref ref-type="fig" rid="fig3">Figures 3A</xref>,<xref ref-type="fig" rid="fig3">B</xref>). The first 2 participants also performed <italic>n<sub>H</sub> =&#x2009;1</italic> correctly, while participant 3 did not learn the correct strategy until executing <italic>n<sub>H</sub> =&#x2009;2.</italic> Note that participants 1 and 2 performed <italic>n<sub>H</sub> =&#x2009;1</italic> before <italic>n<sub>H</sub> =&#x2009;2</italic> and were able to apply what they learned in <italic>n<sub>H</sub> =&#x2009;1</italic> to <italic>n<sub>H</sub> =&#x2009;2.</italic> Because of this, a very fast learning process can be seen during the first <italic>n<sub>H</sub> =&#x2009;2</italic> block. In <xref ref-type="fig" rid="fig3">Figure 3C</xref>, note that some RTs are negative. In these cases, the participant did not wait for the presentation of the GO signal to start the movement.</p>
<fig position="float" id="fig3">
<label>Figure 3</label>
<caption>
<p>Behavioral results for three representative participants. Rows and columns refer to horizons (<italic>n</italic><sub>
<italic>H</italic>
</sub>) and participants, respectively. <bold>(A)</bold> Performance per episode. <bold>(B)</bold> Choice behavior per trial, in terms of selecting the bigger or smaller stimulus. Results are gathered by horizon (<italic>n</italic><sub>
<italic>H</italic>
</sub>) and respective trial within episode (<italic>T</italic><sub>
<italic>E</italic>
</sub>). <bold>(C)</bold> Cumulative density function (CDF) of reaction times. The color code indicates the trial within episode (green for <italic>T</italic><sub>
<italic>E</italic>
</sub> =&#x2009;1, blue for <italic>T</italic><sub>
<italic>E</italic>
</sub> =&#x2009;2, and red for <italic>T</italic><sub>
<italic>E</italic>
</sub> =&#x2009;3). <bold>(D)</bold> Order of execution of blocks and horizons.</p>
</caption>
<graphic xlink:href="fnbeh-18-1399394-g003.tif"/>
</fig>
</sec>
<sec id="sec5">
<label>2.3</label>
<title>A Neurally-inspired model of consequential decision-making</title>
<p>In this section we describe our mathematical formalization of consequential decision-making which incorporates a variable foresight mechanism and adapts to the distribution of reward value across trials. The formalization takes the form of a three-layer neural model. In brief, the bottommost layer is a mean-field model for binary decision-making. The mean-field is driven by a strategy learning layer which then dictate the choices to the decision-making process.</p>
<p>We feel this novel approach yields several advantages over more classical models (i.e., reinforcement learning, drift-diffusion, urgency-gating, etc.). In brief, we aim to provide a formalization of the neural processes involved in reward-driven, delayed-value, multi-step decisions in a context in which attaining reward is contingent on learning the covert effect of actions on the environment. In other words, learning must operate in the absence of explicit performance feedback. Another unique aspect of our approach is the incorporation of a foresight mechanism which adapts to the covert relationship between actions and their effect on the environment as well as to the distribution of reward value across the trials of an episode. We expand on the reasoning behind the creation of our novel formalization in the Discussion section.</p>
<sec id="sec6">
<label>2.3.1</label>
<title>Layer 1: Neural dynamics</title>
<p>To describe the neural dynamics at each trial, we used a mean-field approximation of a biophysically based binary decision-making model (<xref ref-type="bibr" rid="ref94">Wilson and Cowan, 1972</xref>; <xref ref-type="bibr" rid="ref10">Brunel and Wang, 2001</xref>; <xref ref-type="bibr" rid="ref88">Wang, 2002</xref>; <xref ref-type="bibr" rid="ref82">Thura et al., 2022</xref>). This approximation is often used to analyze neuronal dynamics in contexts where mean population activity is relevant. It has been shown that even simple mean-field approximations leveraging as little as two internal variables could reproduce most features of the underlying spiking neuron model (<xref ref-type="bibr" rid="ref96">Wong and Wang, 2006</xref>).</p>
<p>The core of the model consists of two populations of excitatory neurons: one sensitive to the stimulus on the left-hand side of the screen (L), and the other to the stimulus on the right (R). The intensity of the evidence is the size of each stimulus, which is directly proportional to the amount of reward displayed. In the model this is captured by the parameters &#x03BB;<sub>L</sub>, &#x03BB;<sub>R,</sub> respectively. Though distinguishing between the bigger and smaller stimulus values is critical in our task, in the model it is convenient to characterize stimuli based on their position, i.e., left/right. The reason being that the information regarding target size is already conveyed by the respective stimuli values, i.e., the parameters &#x03BB;<sub>L</sub>, &#x03BB;<sub>R</sub>. Moreover, this allows us to introduce an extra degree of freedom in the model without increasing the number of variables. The equations</p>
<disp-formula id="EQ1">
<label>(1)</label>
<mml:math id="M16">
<mml:mrow>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mi>&#x03C4;</mml:mi>
<mml:mfrac>
<mml:mrow>
<mml:mi>d</mml:mi>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mi>L</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:mi>d</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mo>=</mml:mo>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mi>L</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>+</mml:mo>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>&#x03BB;</mml:mi>
<mml:mi>L</mml:mi>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi>&#x03C9;</mml:mi>
<mml:mo>+</mml:mo>
</mml:msub>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mi>L</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>&#x03C9;</mml:mi>
<mml:mo>&#x2212;</mml:mo>
</mml:msub>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mi>R</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>+</mml:mo>
<mml:mi>&#x03C3;</mml:mi>
<mml:msub>
<mml:mi>&#x03BE;</mml:mi>
<mml:mi>L</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mi>&#x03C4;</mml:mi>
<mml:mfrac>
<mml:mrow>
<mml:mi>d</mml:mi>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mi>R</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:mi>d</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mo>=</mml:mo>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mi>R</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>+</mml:mo>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>&#x03BB;</mml:mi>
<mml:mi>R</mml:mi>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi>&#x03C9;</mml:mi>
<mml:mo>+</mml:mo>
</mml:msub>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mi>R</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>&#x03C9;</mml:mi>
<mml:mo>&#x2212;</mml:mo>
</mml:msub>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mi>L</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>+</mml:mo>
<mml:mi>&#x03C3;</mml:mi>
<mml:msub>
<mml:mi>&#x03BE;</mml:mi>
<mml:mi>R</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>describe the temporal dynamics of the firing rates (<italic>r<sub>L</sub>, r<sub>R</sub></italic>) for each of the two populations, and may be interpreted as originating from a neural network as shown in <xref ref-type="fig" rid="fig4">Figure 4A</xref>. Each pool has recurrent excitation (&#x03C9;<sub>+</sub>), and mutual inhibition (&#x03C9;<sub>&#x2212;</sub>). Although the schematic indicates that both excitation and inhibition emanate from a single population of excitatory neurons, this connectivity could be achieved with an equivalent network of excitatory and inhibitory subpopulations (<xref ref-type="bibr" rid="ref96">Wong and Wang, 2006</xref>; <xref ref-type="bibr" rid="ref57">Moreno-Bote et al., 2007</xref>; <xref ref-type="bibr" rid="ref95">Wong et al., 2007</xref>; <xref ref-type="bibr" rid="ref54">Marcos et al., 2013</xref>; <xref ref-type="bibr" rid="ref82">Thura et al., 2022</xref>). In particular, we refer to the work by <xref ref-type="bibr" rid="ref96">Wong and Wang (2006)</xref> in which they reduced a spiking neural network of both excitatory and inhibitory neurons to a two-variable system describing the firing rate of the mean-field dynamics of two populations of excitatory neurons. We opted for this simplified architecture because it is equivalent to the more complex model under certain conditions and provides a more compact formulation. Furthermore, the network shares a basic feature with many other models of bi-stability: to ensure that only one population is active at a time (mutual exclusivity; (<xref ref-type="bibr" rid="ref51">Leopold and Logothetis, 1999</xref>; <xref ref-type="bibr" rid="ref68">Rubin, 2003</xref>)), mutual inhibition is exerted between the two populations (<xref ref-type="bibr" rid="ref7">Blake, 1989</xref>; <xref ref-type="bibr" rid="ref49">Laing and Chow, 2002</xref>; <xref ref-type="bibr" rid="ref93">Wilson, 2003</xref>). The overall neuronal dynamics are regulated by the time constant &#x03C4;, and Gaussian noise &#x03BE; with zero mean and standard deviation &#x03C3;. The sigmoidal function <italic>f</italic> is defined as <inline-formula>
<mml:math id="M17">
<mml:mrow>
<mml:mi>f</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>x</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mi>max</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>/</mml:mo>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>+</mml:mo>
<mml:mi>exp</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>x</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x03B8;</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>/</mml:mo>
<mml:mover accent="true">
<mml:mi>k</mml:mi>
<mml:mo>&#x02DC;</mml:mo>
</mml:mover>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, with <inline-formula>
<mml:math id="M18">
<mml:mrow>
<mml:msub>
<mml:mi>F</mml:mi>
<mml:mrow>
<mml:mi>max</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> denoting the firing rate saturation value.</p>
<fig position="float" id="fig4">
<label>Figure 4</label>
<caption>
<p><bold>(A)</bold> Network structure of binary decision model of mean-field dynamics. The L pool is selective for the stimulus L (&#x03BB;<sub>L</sub>), while the other population is sensitive to the appearance of the stimulus R (&#x03BB;<sub>R</sub>). The two pools mutually inhibit each other (&#x03C9;<sub>&#x2212;</sub>) and have self-excitatory recurrent connections (&#x03C9;<sub>+</sub>). <bold>(B)</bold> Firing rate of the two populations (L, R) of excitatory neurons according to the dynamics in <xref ref-type="disp-formula" rid="EQ1">Eq. 1</xref>. A decision is taken at time 506&#x2009;ms (vertical dashed line) when the difference in activity between L and R pools passes the threshold of &#x0394; =25&#x2009;Hz. The strengths of the stimuli are set to <italic>&#x03BB;</italic><sub>L</sub> =&#x2009;0.0203 and <italic>&#x03BB;</italic><sub>R</sub> =&#x2009;0.0227. The time constant and the noise are set to <italic>&#x03C4;</italic> =&#x2009;80&#x2009;ms and <italic>&#x03C3;</italic> =&#x2009;0.003&#x2009;ms<sup>&#x2212;1</sup>, respectively.</p>
</caption>
<graphic xlink:href="fnbeh-18-1399394-g004.tif"/>
</fig>
<p>The neural dynamics described in this section refer to the time-course of a single trial, and are related to the discrimination of the two stimuli. The model commits to a perceptual decision when the difference between the L and R pool activity crosses a threshold &#x0394; (<xref ref-type="bibr" rid="ref67">Roxin and Ledberg, 2008</xref>), see <xref ref-type="fig" rid="fig4">Figure 4B</xref>. This event defines the trial&#x2019;s decision time. Note that the decision time and the likelihood of picking the larger stimulus are conditioned on the evidence associated with the two stimuli (&#x03BB;<sub>L</sub>, &#x03BB;<sub>R</sub>), i.e., how easy it is to distinguish between them. The larger the difference between the stimuli, the more likely, and quickly, the larger stimulus is selected.</p>
<p>This type of decision-making model is made such that the larger stimulus is always favored. Indeed, according to <xref ref-type="disp-formula" rid="EQ1">Eq. 1</xref>, the target with the stronger evidence is the most likely to be selected. As described in the next section, the addition of the middle layer of our model provides a generalization of this mechanism by allowing the choice between the smaller and the larger target.</p>
</sec>
<sec id="sec7">
<label>2.3.2</label>
<title>Layer 2: Intended decision</title>
<p>While most decision-making models consider only one-shot decisions (<xref ref-type="bibr" rid="ref96">Wong and Wang, 2006</xref>; <xref ref-type="bibr" rid="ref67">Roxin and Ledberg, 2008</xref>; <xref ref-type="bibr" rid="ref69">Salinas, 2008</xref>; <xref ref-type="bibr" rid="ref33">Hern&#x00E1;ndez et al., 2010</xref>; <xref ref-type="bibr" rid="ref42">Kilpatrick et al., 2019</xref>), the increased temporal span and the various sources of uncertainty inherent in the consequential task necessitate the addition of a layer to the model. The second layer of the model enables dynamic shifting between the natural impulse to choose the larger stimulus and inhibition. We implemented such a mechanism by means of an inhibitory control pool, which regulates the reversal of the selection criterion toward the smaller or larger stimulus. We called this mechanism <italic>intended decision</italic>, as it defines the intended target to select at each trial. This layer enables the model to switch preference as a function of context (see layer 3 description).</p>
<p>The intended decision mechanism is represented by a two-attractor dynamical system. The state of the model can be interpreted as the continuous expression of the tendency to select one choice over another. The attractors are the states toward which the dynamics of the system naturally evolve. Since we have two choices, we considered the energy function <inline-formula>
<mml:math id="M19">
<mml:mrow>
<mml:mi>E</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>&#x03C8;</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:msup>
<mml:mi>&#x03C8;</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:msup>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>&#x03C8;</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> which has two basins of attraction at 0 and 1. The basins at 0 and 1 are associated with the small and big stimulus, respectively (see <xref ref-type="fig" rid="fig5">Figure 5A</xref>). Hence, the dynamics of <italic>&#x03C8;</italic> are determined by</p>
<disp-formula id="EQ2">
<label>(2)</label>
<mml:math id="M20">
<mml:mrow>
<mml:msub>
<mml:mi>&#x03C4;</mml:mi>
<mml:mi>&#x03C8;</mml:mi>
</mml:msub>
<mml:mfrac>
<mml:mrow>
<mml:mi>d</mml:mi>
<mml:mi>&#x03C8;</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mrow>
<mml:mi>d</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mo>=</mml:mo>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>4</mml:mn>
<mml:mi>&#x03C8;</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>&#x03C8;</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>&#x03C8;</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
<mml:mo>/</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>+</mml:mo>
<mml:mfrac>
<mml:mn>1</mml:mn>
<mml:mrow>
<mml:msup>
<mml:mi>t</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:mfrac>
<mml:msub>
<mml:mi>&#x03C3;</mml:mi>
<mml:mi>&#x03C8;</mml:mi>
</mml:msub>
<mml:msub>
<mml:mi>&#x03BE;</mml:mi>
<mml:mi>&#x03C8;</mml:mi>
</mml:msub>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>where &#x03C4;<italic><sub>&#x03C8;</sub></italic> is a time constant. The Gaussian noise <italic>&#x03BE;<sub>&#x03C8;</sub>(t)</italic> is scaled by a constant (<italic>&#x03C3;<sub>&#x03C8;</sub></italic>) and decays quadratically with time. Thus, the noise exerts a strong influence at the beginning of the process and becomes increasingly negligible as the system approaches either of the basins.</p>
<fig position="float" id="fig5">
<label>Figure 5</label>
<caption>
<p>Dynamics of the second layer of the model. <bold>(A)</bold> Energy function <inline-formula>
<mml:math id="M21">
<mml:mrow>
<mml:mi>E</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>&#x03C8;</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:msup>
<mml:mi>&#x03C8;</mml:mi>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:msup>
<mml:mrow>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>&#x03C8;</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> with two basins of attraction at 0 and 1, associated with the small/big targets, respectively. The small circle represents a possible initial condition for the dynamics of <inline-formula>
<mml:math id="M22">
<mml:mi>&#x03C8;</mml:mi>
</mml:math>
</inline-formula>. <bold>(B)</bold> Ten simulated trajectories for <inline-formula>
<mml:math id="M23">
<mml:mrow>
<mml:mi>&#x03C8;</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> according to <xref ref-type="disp-formula" rid="EQ2">Eq. 2</xref> with initial condition <inline-formula>
<mml:math id="M24">
<mml:mrow>
<mml:mi>&#x03C8;</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:mn>0.45</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> and noise amplitude &#x03C3;<sub>&#x03C8;</sub> =&#x2009;0.4&#x2009;ms<sup>&#x2212;1</sup>.</p>
</caption>
<graphic xlink:href="fnbeh-18-1399394-g005.tif"/>
</fig>
<p>If we set the initial condition to <inline-formula>
<mml:math id="M25">
<mml:mrow>
<mml:msub>
<mml:mi>&#x03C8;</mml:mi>
<mml:mn>0</mml:mn>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mn>0.5</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> and let the system evolve, the final state would be either 0 or 1 with equal probability. Shifting the initial condition toward one of the attractors results in an increased probability of the system ending in the corresponding basin, and ultimately its fixed point. <xref ref-type="fig" rid="fig5">Figure 5B</xref> shows 10 simulated trajectories of <inline-formula>
<mml:math id="M26">
<mml:mrow>
<mml:mi>&#x03C8;</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> where the initial condition was set to <inline-formula>
<mml:math id="M27">
<mml:mrow>
<mml:msub>
<mml:mi>&#x03C8;</mml:mi>
<mml:mn>0</mml:mn>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mn>0.45</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>. Since the initial condition is smaller than 0.5, most of the trajectories reach the fixed point at 0 and only a few of them, due to the initial noise, reach 1 as their final state.</p>
<p>The initial condition (<inline-formula>
<mml:math id="M28">
<mml:mrow>
<mml:msub>
<mml:mi>&#x03C8;</mml:mi>
<mml:mn>0</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>) and the noise intensity (<italic>&#x03C3;<sub>&#x03C8;</sub></italic>) are interdependent. The closer an initial condition is to one of the attractors, the larger the noise must be to escape the corresponding basin of attraction. Behaviorally, the role of the initial condition is to capture the <italic>a-priori</italic> bias of choosing the smaller/bigger target. Please note, however, that a strong initial bias toward one of the targets does not guarantee the final decision, especially when the level of uncertainty is large. Because of this behavioral effect, we refer to the noise intensity <italic>&#x03C3;<sub>&#x03C8;</sub></italic> as <italic>decisional uncertainty.</italic></p>
<p>The evolution of the dynamical system in <xref ref-type="disp-formula" rid="EQ2">Eq. 2</xref> describes the intention of the decision-making process, at each trial <italic>T</italic>, to choose the smaller/bigger target. The intention is established once a fixed point is reached. We call <inline-formula>
<mml:math id="M29">
<mml:mrow>
<mml:mover accent="true">
<mml:mi>&#x03C8;</mml:mi>
<mml:mo>&#x02DC;</mml:mo>
</mml:mover>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>T</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> the fixed point reached at trial <italic>T</italic>, i.e.,</p>
<disp-formula id="E1">
<mml:math id="M30">
<mml:mrow>
<mml:mover accent="true">
<mml:mi>&#x03C8;</mml:mi>
<mml:mo>&#x02DC;</mml:mo>
</mml:mover>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>T</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:munder>
<mml:mrow>
<mml:mi>lim</mml:mi>
</mml:mrow>
<mml:mrow>
<mml:mi>t</mml:mi>
<mml:mo>&#x2192;</mml:mo>
<mml:mi>&#x221E;</mml:mi>
</mml:mrow>
</mml:munder>
<mml:mi>&#x03C8;</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>t</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:mtable>
<mml:mtr>
<mml:mtd>
<mml:mn>0</mml:mn>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mn>1</mml:mn>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mrow>
</mml:mrow>
</mml:math>
</disp-formula>
<p>is the intended decision of choosing the smaller (0) or bigger (1) stimulus.</p>
<p>Although the small/big stimulus may be favored at each trial, the final decision still depends on the stimuli intensity ratio. More specifically, if the evidence associated with the small/large stimulus is higher/lower than that of its counterpart, the dynamics of the system will evolve as described in the previous section, see <xref ref-type="disp-formula" rid="EQ1">Eq. 1</xref>. For this reason, we incorporated the <italic>intention</italic> term <inline-formula>
<mml:math id="M31">
<mml:mrow>
<mml:mover accent="true">
<mml:mi>&#x03C8;</mml:mi>
<mml:mo>&#x02DC;</mml:mo>
</mml:mover>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>T</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> in <xref ref-type="disp-formula" rid="EQ1">Eq. 1</xref> which connects the <italic>intended decision layer</italic> with the <italic>neural dynamics layer</italic>. This yields a novel set of equations</p>
<disp-formula id="EQ3">
<label>(3)</label>
<mml:math id="M32">
<mml:mrow>
<mml:mfenced close="" open="{">
<mml:mrow>
<mml:mtable equalrows="true" equalcolumns="true">
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mi>&#x03C4;</mml:mi>
<mml:mfrac>
<mml:mrow>
<mml:mi>d</mml:mi>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mi>L</mml:mi>
</mml:msub>
<mml:mfenced>
<mml:mi>t</mml:mi>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mi>d</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mo>=</mml:mo>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mi>L</mml:mi>
</mml:msub>
<mml:mfenced>
<mml:mi>t</mml:mi>
</mml:mfenced>
<mml:mo>+</mml:mo>
<mml:mi>f</mml:mi>
<mml:mfenced>
<mml:mtable columnalign="left">
<mml:mtr>
<mml:mtd>
<mml:mover accent="true">
<mml:mi>&#x03C8;</mml:mi>
<mml:mo>&#x02DC;</mml:mo>
</mml:mover>
<mml:mfenced>
<mml:mi mathvariant="normal">T</mml:mi>
</mml:mfenced>
<mml:msub>
<mml:mi>&#x03BB;</mml:mi>
<mml:mi>L</mml:mi>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:mfenced>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mover accent="true">
<mml:mi>&#x03C8;</mml:mi>
<mml:mo>&#x02DC;</mml:mo>
</mml:mover>
<mml:mfenced>
<mml:mi>T</mml:mi>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:msub>
<mml:mi>&#x03BB;</mml:mi>
<mml:mi>R</mml:mi>
</mml:msub>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi>&#x03C9;</mml:mi>
<mml:mo>+</mml:mo>
</mml:msub>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mi>L</mml:mi>
</mml:msub>
<mml:mfenced>
<mml:mi>t</mml:mi>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>&#x03C9;</mml:mi>
<mml:mo>&#x2212;</mml:mo>
</mml:msub>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mi>R</mml:mi>
</mml:msub>
<mml:mfenced>
<mml:mi>t</mml:mi>
</mml:mfenced>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mfenced>
<mml:mo>+</mml:mo>
<mml:mi>&#x03C3;</mml:mi>
<mml:msub>
<mml:mi>&#x03BE;</mml:mi>
<mml:mi>L</mml:mi>
</mml:msub>
<mml:mfenced>
<mml:mi>t</mml:mi>
</mml:mfenced>
</mml:mrow>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mrow>
<mml:mi>&#x03C4;</mml:mi>
<mml:mfrac>
<mml:mrow>
<mml:mi>d</mml:mi>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mi>R</mml:mi>
</mml:msub>
<mml:mfenced>
<mml:mi>t</mml:mi>
</mml:mfenced>
</mml:mrow>
<mml:mrow>
<mml:mi>d</mml:mi>
<mml:mi>t</mml:mi>
</mml:mrow>
</mml:mfrac>
<mml:mo>=</mml:mo>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mi>R</mml:mi>
</mml:msub>
<mml:mfenced>
<mml:mi>t</mml:mi>
</mml:mfenced>
<mml:mo>+</mml:mo>
<mml:mi>f</mml:mi>
<mml:mfenced>
<mml:mtable columnalign="left">
<mml:mtr>
<mml:mtd>
<mml:mover accent="true">
<mml:mi>&#x03C8;</mml:mi>
<mml:mo>&#x02DC;</mml:mo>
</mml:mover>
<mml:mfenced>
<mml:mi mathvariant="normal">T</mml:mi>
</mml:mfenced>
<mml:msub>
<mml:mi>&#x03BB;</mml:mi>
<mml:mi>R</mml:mi>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:mfenced>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mover accent="true">
<mml:mi>&#x03C8;</mml:mi>
<mml:mo>&#x02DC;</mml:mo>
</mml:mover>
<mml:mfenced>
<mml:mi>T</mml:mi>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:msub>
<mml:mi>&#x03BB;</mml:mi>
<mml:mi>L</mml:mi>
</mml:msub>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mo>+</mml:mo>
<mml:msub>
<mml:mi>&#x03C9;</mml:mi>
<mml:mo>+</mml:mo>
</mml:msub>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mi>R</mml:mi>
</mml:msub>
<mml:mfenced>
<mml:mi>t</mml:mi>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>&#x03C9;</mml:mi>
<mml:mo>&#x2212;</mml:mo>
</mml:msub>
<mml:msub>
<mml:mi>r</mml:mi>
<mml:mi>L</mml:mi>
</mml:msub>
<mml:mfenced>
<mml:mi>t</mml:mi>
</mml:mfenced>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mfenced>
<mml:mo>+</mml:mo>
<mml:mi>&#x03C3;</mml:mi>
<mml:msub>
<mml:mi>&#x03BE;</mml:mi>
<mml:mi>R</mml:mi>
</mml:msub>
<mml:mfenced>
<mml:mi>t</mml:mi>
</mml:mfenced>
</mml:mrow>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:math>
</disp-formula>
<p>which is able to switch preferences between the large and small stimulus. If <inline-formula>
<mml:math id="M33">
<mml:mrow>
<mml:mover accent="true">
<mml:mi>&#x03C8;</mml:mi>
<mml:mo>&#x02DC;</mml:mo>
</mml:mover>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>T</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>, the larger stimulus is favored (and the equations reduce to <xref ref-type="disp-formula" rid="EQ1">Eq. 1</xref>); however, if <inline-formula>
<mml:math id="M34">
<mml:mrow>
<mml:mover accent="true">
<mml:mi>&#x03C8;</mml:mi>
<mml:mo>&#x02DC;</mml:mo>
</mml:mover>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>T</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> the smaller stimulus is preferred.</p>
<p>In summary, the <italic>intended decision</italic> layer enables the model to dynamically adjust preferences for the bigger or smaller stimulus. This inhibitory control plays the role of the regulatory criterion (size-wise) with which a decision is made in the consequential task, as described by <xref ref-type="disp-formula" rid="EQ2">Eq. 2</xref>.</p>
</sec>
<sec id="sec8">
<label>2.3.3</label>
<title>Layer 3: Learning the strategy</title>
<p>Although the previously described intended decision layer enabled the model to target a specific type of stimulus at each trial, a second mechanism is required to internally oversee performance and to promote beneficial strategies. In the consequential task, the goal is to maximize the cumulative reward value obtained in each episode. As shown in previous analyses, most participants learned the optimal strategy after an exploratory phase, gradually improving their performance until the optimum was reached. Inspired by the same principle of exploration and reinforcement, we incorporated the strategy learning layer in our model.</p>
<p>The internal dynamics of an episode are such that selecting the small/large stimulus in a trial results in an increase/decrease of the mean value of the presented stimuli in the next trial (<xref ref-type="fig" rid="fig1">Figure 1</xref>). Consequently, the strategy to maximize the reward value must vary as a function of trial within episode (<italic>T<sub>E</sub></italic>). For clarity, each trial <italic>T</italic> is associated with an episode <italic>E</italic> and number of trial within episode <italic>T<sub>E</sub></italic>. We use both notations interchangeably, i.e., <italic>T&#x2009;=</italic> (<italic>E, T<sub>E</sub></italic>).</p>
<p>The strategy learning mechanism in the model reinforces beneficial strategies and weakens less rewarding ones, see Discussion for a comparison with existing models. Following each episode <italic>E,</italic> the strategy function <inline-formula>
<mml:math id="M35">
<mml:mrow>
<mml:mi>&#x03D5;</mml:mi>
<mml:mo>=</mml:mo>
<mml:mi>&#x03D5;</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>E</mml:mi>
<mml:mi mathvariant="normal">,</mml:mi>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>E</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> is updated by considering the intended choice <inline-formula>
<mml:math id="M36">
<mml:mrow>
<mml:mover accent="true">
<mml:mi>&#x03C8;</mml:mi>
<mml:mo>&#x02DC;</mml:mo>
</mml:mover>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>T</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> and the obtained reward value <italic>R(T)</italic>. In our case, reward value originates from each participant&#x2019;s subjective evaluation in the absence of explicit performance feedback. This internal assessment yields a positive or negative perception of reward, i.e., a subjective reward. Learning implies that the preference for the selected strategy is reinforced if the participant&#x2019;s internal assessment results in positive subjective reward. Namely, with a positive reward (<italic>R(T)&#x2009;&#x003E;&#x2009;0</italic>), <inline-formula>
<mml:math id="M37">
<mml:mi>&#x03D5;</mml:mi>
</mml:math>
</inline-formula> is increased if the larger stimulus was chosen (<inline-formula>
<mml:math id="M38">
<mml:mrow>
<mml:mover accent="true">
<mml:mi>&#x03C8;</mml:mi>
<mml:mo>&#x02DC;</mml:mo>
</mml:mover>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>T</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>) and decreased otherwise (<inline-formula>
<mml:math id="M39">
<mml:mrow>
<mml:mover accent="true">
<mml:mi>&#x03C8;</mml:mi>
<mml:mo>&#x02DC;</mml:mo>
</mml:mover>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>T</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>). Notice that a negative reward discourages the current strategy but promotes the exploration of alternative strategies and makes it possible to learn the optimal one over time. Mathematically, we describe the dynamics of learning as</p>
<disp-formula id="EQ4">
<label>(4)</label>
<mml:math id="M40">
<mml:mtable columnalign="left">
<mml:mtr>
<mml:mtd>
<mml:mi>&#x03D5;</mml:mi>
<mml:mfenced>
<mml:mrow>
<mml:mi>E</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
<mml:mi mathvariant="normal">,</mml:mi>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>E</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>=</mml:mo>
<mml:mi>&#x03D5;</mml:mi>
<mml:mfenced>
<mml:mrow>
<mml:mi>E</mml:mi>
<mml:mi mathvariant="normal">,</mml:mi>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>E</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mtd>
</mml:mtr>
<mml:mtr>
<mml:mtd>
<mml:mo>+</mml:mo>
<mml:mi>k</mml:mi>
<mml:mi>R</mml:mi>
<mml:mfenced>
<mml:mrow>
<mml:mi>E</mml:mi>
<mml:mi mathvariant="normal">,</mml:mi>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>E</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mfenced>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mover accent="true">
<mml:mi>&#x03C8;</mml:mi>
<mml:mo>&#x02DC;</mml:mo>
</mml:mover>
<mml:mfenced>
<mml:mrow>
<mml:mi>E</mml:mi>
<mml:mi mathvariant="normal">,</mml:mi>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>E</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
<mml:msup>
<mml:mfenced>
<mml:mrow>
<mml:mi>&#x03D5;</mml:mi>
<mml:mfenced>
<mml:mrow>
<mml:mi>E</mml:mi>
<mml:mi mathvariant="normal">,</mml:mi>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>E</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:msup>
<mml:mfenced>
<mml:mrow>
<mml:mi>&#x03D5;</mml:mi>
<mml:mfenced>
<mml:mrow>
<mml:mi>E</mml:mi>
<mml:mi mathvariant="normal">,</mml:mi>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>E</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mtd>
</mml:mtr>
</mml:mtable>
</mml:math>
</disp-formula>
<p>where <italic>k</italic> is the learning rate. Note that if <italic>k&#x2009;=&#x2009;0</italic>, <inline-formula>
<mml:math id="M41">
<mml:mrow>
<mml:mi>&#x03D5;</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>E</mml:mi>
<mml:mi mathvariant="normal">,</mml:mi>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>E</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> remains constant and, therefore, there is no learning. The term <inline-formula>
<mml:math id="M42">
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mfenced>
<mml:mrow>
<mml:mi>&#x03D5;</mml:mi>
<mml:mfenced>
<mml:mrow>
<mml:mi>E</mml:mi>
<mml:mi mathvariant="normal">,</mml:mi>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>E</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
<mml:msup>
<mml:mrow>
<mml:mfenced>
<mml:mrow>
<mml:mi>&#x03D5;</mml:mi>
<mml:mfenced>
<mml:mrow>
<mml:mi>E</mml:mi>
<mml:mi mathvariant="normal">,</mml:mi>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>E</mml:mi>
</mml:msub>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
</mml:mfenced>
</mml:mrow>
<mml:mn>2</mml:mn>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula> is required to gradually reduce the increment to zero the closer <inline-formula>
<mml:math id="M43">
<mml:mi>&#x03D5;</mml:mi>
</mml:math>
</inline-formula> gets to either zero or one. This bounds <inline-formula>
<mml:math id="M44">
<mml:mi>&#x03D5;</mml:mi>
</mml:math>
</inline-formula> to the interval [0, 1]. The reward function <italic>R(</italic><inline-formula>
<mml:math id="M45">
<mml:mrow>
<mml:mi>E</mml:mi>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>E</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula><italic>)</italic> represents the subjective reward. The only requirement for this function is that <italic>R</italic>(<inline-formula>
<mml:math id="M46">
<mml:mrow>
<mml:mi>E</mml:mi>
<mml:mo>,</mml:mo>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>E</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>) must be positive or negative if the subjective reward is considered beneficial or not, respectively. In the case of the current task, participants must look for clues that convey indirect information about their performance. The key observation participants had to make was the change in stimuli mean <italic>M</italic> between consecutive trials in an episode as a result of their choices. For this reason, we defined the reward function as <inline-formula>
<mml:math id="M47">
<mml:mrow>
<mml:mi>R</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>E</mml:mi>
<mml:mi mathvariant="normal">,</mml:mi>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>E</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:mi>M</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>E</mml:mi>
<mml:mi mathvariant="normal">,</mml:mi>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>E</mml:mi>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>M</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>E</mml:mi>
<mml:mi mathvariant="normal">,</mml:mi>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>E</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> (see <xref ref-type="disp-formula" rid="EQ4">Eq. 4</xref>). We discuss how the reward function could generalize to different tasks in the conclusions section.</p>
<p>The strategy layer operates a longer time scale than the lower layers. The strategy is updated at the end of each episode by reinforcing/weakening the policy that has yielded a positive/negative reward. Mathematically, as mentioned before, this means that with a positive reward (<italic>R(T)&#x2009;&#x003E;&#x2009;0</italic>), <inline-formula>
<mml:math id="M48">
<mml:mi>&#x03D5;</mml:mi>
</mml:math>
</inline-formula> is increased if the larger stimulus was chosen (<inline-formula>
<mml:math id="M49">
<mml:mrow>
<mml:mover accent="true">
<mml:mi>&#x03C8;</mml:mi>
<mml:mo>&#x02DC;</mml:mo>
</mml:mover>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>T</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>) and decreased otherwise (<inline-formula>
<mml:math id="M50">
<mml:mrow>
<mml:mover accent="true">
<mml:mi>&#x03C8;</mml:mi>
<mml:mo>&#x02DC;</mml:mo>
</mml:mover>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>T</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>). In the case that both the larger stimulus is repeatedly chosen and positive rewards are obtained, then <inline-formula>
<mml:math id="M51">
<mml:mi>&#x03D5;</mml:mi>
</mml:math>
</inline-formula> converges to 1. In contrast, if both the smaller stimulus is repeatedly chosen and positive rewards are obtained, then <inline-formula>
<mml:math id="M52">
<mml:mi>&#x03D5;</mml:mi>
</mml:math>
</inline-formula> converges to 0. This update manifests as a change in the initial condition for the intended decision <inline-formula>
<mml:math id="M53">
<mml:mi>&#x03C8;</mml:mi>
</mml:math>
</inline-formula> (<xref ref-type="disp-formula" rid="EQ2">Eq. 2</xref>), i.e., biasing the direction, small or big, for the intended decision to go. As shown in <xref ref-type="fig" rid="fig5">Figure 5</xref>, shifting the initial condition toward one of the two basins (0 or 1) increases the probability of reaching it. Mathematically, this can be implemented by setting <inline-formula>
<mml:math id="M54">
<mml:mrow>
<mml:mi>&#x03C8;</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mn>0</mml:mn>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:mi>&#x03D5;</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mi>T</mml:mi>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> for each trial. In this way, the connection between the intended decision and strategy layers lies in the influence the strategy learning exerts at each decision.</p>
<p>To conclude, our model consists of a three layer structure. The dynamics of each layer are defined by <xref ref-type="disp-formula" rid="EQ3">Eq. 3</xref> (neural dynamics), <xref ref-type="disp-formula" rid="EQ2">Eq. 2</xref> (intended decision), and <xref ref-type="disp-formula" rid="EQ4">Eq. 4</xref> (strategy learning). <xref ref-type="fig" rid="fig6">Figure 6</xref> shows a schematic of the complete model. The bottom part depicts the neural dynamics originating from two pools of neurons which encode the responses to two external stimuli (<italic>L, R</italic>). The middle shows the intended decision layer at every trial. Finally, the top is the strategy learning layer which evolves at a much slower timescale; the combined information of the intended decision and the subjective reward drives strategy learning.</p>
<fig position="float" id="fig6">
<label>Figure 6</label>
<caption>
<p>Multi-layer network structure of mean-field model of consequence-based decision making, in the case of a horizon 1 experiment. From the bottom: Neural dynamics layer: pool L is selective for stimulus L (&#x03BB;<sub>L</sub>), while the other population is sensitive to the appearance of stimulus R (&#x03BB;<sub>R</sub>). The two pools mutually inhibit each other (&#x03C9;<sub>&#x2212;</sub>) and have self-excitatory recurrent connections (&#x03C9;<sub>+</sub>). The dynamics of the firing rate of the two populations is regulated by <xref ref-type="disp-formula" rid="EQ3">Eq. 3</xref>. Intended decision layer: the function &#x03C8; represents the intention, in terms of decision process, made at each trial T, of aiming for the smaller or bigger target. The dynamics of the intended decision is regulated by <xref ref-type="disp-formula" rid="EQ2">Eq. 2</xref>. Strategy learning layer: after each trial the strategy is revised, in a reinforcement learning fashion, depending on the magnitude of the gained reward value. The strategy is updated according to <xref ref-type="disp-formula" rid="EQ4">Eq. 4</xref>.</p>
</caption>
<graphic xlink:href="fnbeh-18-1399394-g006.tif"/>
</fig>
</sec>
</sec>
<sec id="sec9">
<label>2.4</label>
<title>Model simulations</title>
<p>We performed a parameter space analysis to assess the influence of the model parameters on the main behavioral metrics of interest: reaction time (RT) and performance (PF). To obtain meaningful biophysical results for the neuronal dynamics, we simulated our model varying the time constant &#x03C4;, the noise amplitude &#x03C3;, and the decision threshold &#x0394; (in <xref ref-type="disp-formula" rid="EQ3">Eq. 3</xref>) in the following ranges: <inline-formula>
<mml:math id="M55">
<mml:mrow>
<mml:mi>&#x03C4;</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mrow>
<mml:mo>[</mml:mo>
<mml:mrow>
<mml:mn>25</mml:mn>
<mml:mi mathvariant="normal">,</mml:mi>
<mml:mn>95</mml:mn>
</mml:mrow>
<mml:mo>]</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> <italic>ms</italic>, <inline-formula>
<mml:math id="M56">
<mml:mrow>
<mml:mi>&#x03C3;</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mrow>
<mml:mo>[</mml:mo>
<mml:mrow>
<mml:msup>
<mml:mrow>
<mml:mn>10</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:msup>
<mml:mi mathvariant="normal">,</mml:mi>
<mml:msup>
<mml:mrow>
<mml:mn>10</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
<mml:mo>]</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> <italic>ms<sup>&#x2212;1</sup></italic>, and <inline-formula>
<mml:math id="M57">
<mml:mrow>
<mml:mi>&#x0394;</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mrow>
<mml:mo>[</mml:mo>
<mml:mrow>
<mml:mn>0.01</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0.035</mml:mn>
</mml:mrow>
<mml:mo>]</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> <italic>ms<sup>&#x2212;1</sup></italic> (see (<xref ref-type="bibr" rid="ref54">Marcos et al., 2013</xref>). Also, we set F<sub>max</sub>&#x2009;=&#x2009;0.04&#x2009;<italic>ms<sup>&#x2212;1</sup></italic>, &#x03B8;&#x2009;=&#x2009;0.015&#x2009;<italic>ms<sup>&#x2212;1</sup></italic>, <inline-formula>
<mml:math id="M58">
<mml:mover accent="true">
<mml:mi>k</mml:mi>
<mml:mo>&#x02DC;</mml:mo>
</mml:mover>
</mml:math>
</inline-formula> = 0.022&#x2009;<italic>ms<sup>&#x2212;1</sup></italic>, &#x03C9;<sub>+</sub>&#x2009;=&#x2009;1.4, &#x03C9;<sub>&#x2212;</sub>&#x2009;=&#x2009;1.5. We fixed the parameters defined in the function <italic>f</italic> (see <xref ref-type="disp-formula" rid="EQ3">Eq. 3</xref>) as well as the connection strengths between pools of neurons (&#x03C9;<sub>+</sub> and &#x03C9;<sub>&#x2212;</sub>), as in (<xref ref-type="bibr" rid="ref54">Marcos et al., 2013</xref>). As we will see below, by only varying &#x03C4;, &#x03C3;, and &#x0394; we can simulate a wide range of different behaviors. In <xref ref-type="disp-formula" rid="EQ2">Eq. 2</xref>, we set &#x03C4;<italic>
<sub>&#x03C8;</sub>
</italic>&#x2009;=&#x2009;10&#x2009;<italic>ms</italic> such that the dynamics of <xref ref-type="disp-formula" rid="EQ2">Eq. 2</xref> is faster than the dynamics of <xref ref-type="disp-formula" rid="EQ3">Eq. 3</xref> while remaining the same order of magnitude. <xref ref-type="fig" rid="fig7">Figure 7A</xref> shows how RT is affected by &#x03C4; and &#x0394;. By increasing the time constant &#x03C4;, the RT increases both in mean and standard deviation (see <xref rid="SM1" ref-type="supplementary-material">Supplementary Figures S4a,d</xref>). The same trend occurs when increasing the threshold &#x0394; (<xref rid="SM1" ref-type="supplementary-material">Supplementary Figures S4b,e</xref>). When varying the noise &#x03C3;, we did not find a substantial difference in the RT (<xref rid="SM1" ref-type="supplementary-material">Supplementary Figures S4c,f</xref>). By fixing &#x03C4;, &#x03C3;, and &#x0394;, we quantified the influence of the learning rate <italic>k</italic> and the decisional uncertainty <italic>&#x03C3;<sub>&#x03C8;</sub></italic> on the PF, and, consequently, on the learning time <italic>t<sub>L</sub></italic> (defined as in section Behavioral Results). <xref ref-type="fig" rid="fig7">Figure 7B</xref> shows that learning time decreases as learning rate <italic>k</italic> increases and decisional uncertainty <italic>&#x03C3;<sub>&#x03C8;</sub></italic> decreases. Note that for these simulations we used <italic>n<sub>H</sub> =&#x2009;1</italic> with 50 episodes, therefore any <italic>t<sub>L</sub></italic> bigger than 50 means the optimal strategy was not learned.</p>
<fig position="float" id="fig7">
<label>Figure 7</label>
<caption>
<p>Parameter space analysis. <bold>(A)</bold> The RT increases when increasing either &#x03C4; or &#x0394; (<italic>&#x03C3;</italic>&#x2009;=&#x2009;0.001&#x2009;ms<sup>&#x2212;1</sup>). <bold>(B)</bold> Learning time (t<sub>L</sub>) decreases when learning rate k increases and when decisional uncertainty decreases &#x03C3;<sub>&#x03C8;</sub> (<italic>&#x03C4;</italic> =&#x2009;81&#x2009;ms, &#x03C3; =&#x2009;0.001&#x2009;ms<sup>&#x2212;1</sup>, and &#x0394; =&#x2009;30&#x2009;Hz).</p>
</caption>
<graphic xlink:href="fnbeh-18-1399394-g007.tif"/>
</fig>
<p>To demonstrate the behavior of the model, <xref ref-type="fig" rid="fig8">Figure 8</xref> shows the results of a typical simulation of a horizon <italic>n<sub>H</sub> =&#x2009;1</italic> experiment. <xref ref-type="fig" rid="fig8">Figure 8A</xref> shows the dynamics of the neural dynamics layer of our model together with the stimuli used in the simulation during the first three episodes. The bottom row shows the time course of the two population firing rates (<xref ref-type="disp-formula" rid="EQ3">Eq. 3</xref>) encoding the stimuli L, R (depicted in the top row). To better understand the progression of this process over time, <xref ref-type="fig" rid="fig8">Figure 8B</xref> provides a view of 36 episodes. The top row shows the performance and difficulty (in terms of difference between stimuli &#x0394;S) metrics. Note that the optimal strategy in this simulation was learned and applied from the 17<sup>th</sup> episode onward. After this point, only the most difficult trials (smallest &#x0394;S) managed to diminish the performance. The same conclusions can be drawn by looking at the time course of the intended decision metric (middle inset). After the 17th episode the intended decision metric exhibits a repeating pattern (small for <italic>T<sub>E</sub> =&#x2009;1</italic>, and big for <italic>T<sub>E</sub> =&#x2009;2</italic>). The bottom row shows the strategy learning. For the first trial within episode (<italic>T<sub>E</sub> =&#x2009;1</italic>), <italic>&#x03D5;</italic> tends to 0, i.e., it pushes the intended decision to choose the smaller stimulus. For the second trial within episode (<italic>T<sub>E</sub> =&#x2009;2</italic>), the trend is reversed, effectively capturing the optimal policy.</p>
<fig position="float" id="fig8">
<label>Figure 8</label>
<caption>
<p>Model example simulations for a horizon 1 block. <bold>(A)</bold> Simulation of the first 3 episodes. Top row: Stimuli presentation with selections indicated by a yellow dot. Bottom row: firing rate of the two populations of neurons encoding the left (in blue) and right (in red) stimuli (<xref ref-type="disp-formula" rid="EQ3">Eq. 3</xref>). Vertical dashed bars indicate the time the decision threshold was crossed. <bold>(B)</bold> Simulation of 36 consecutive episodes. First row: Performance (blue - solid) and difference between stimuli &#x0394;S (green - dashed). Second row: intended decision dynamics of choosing the bigger (1) or smaller (0) stimulus. Third row: evolution of strategy learning for each trial within episode (<italic>T</italic><sub>
<italic>E</italic>
</sub>). Parameters used for the simulations: <italic>G</italic> =&#x2009;0.3, &#x0394; =&#x2009;25&#x2009;Hz, <italic>&#x03C4;</italic> =&#x2009;80&#x2009;ms, <italic>&#x03C3;</italic> =&#x2009;0.006&#x2009;ms<sup>&#x2212;1</sup>, <inline-formula>
<mml:math id="M59">
<mml:mrow>
<mml:msub>
<mml:mi>&#x03D5;</mml:mi>
<mml:mn>0</mml:mn>
</mml:msub>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mi mathvariant="normal">,</mml:mi>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>E</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:mn>0.5</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> for <italic>T</italic><sub>
<italic>E</italic>
</sub> =&#x2009;1,2, <italic>k</italic> =&#x2009;0.4, &#x03C3;<sub>&#x03C8;</sub> =&#x2009;0.4&#x2009;ms<sup>&#x2212;1</sup>.</p>
</caption>
<graphic xlink:href="fnbeh-18-1399394-g008.tif"/>
</fig>
</sec>
<sec id="sec10">
<label>2.5</label>
<title>Individual participants&#x2019; behavioral fit</title>
<p>In this section we describe the fit of the model parameters to the participants&#x2019; individual behavioral metrics. The first step is to find the best fit for the neural dynamics by fitting the reaction time (RT) and the visual discrimination (VD), i.e., fit the parameters involved in <xref ref-type="disp-formula" rid="EQ3">Eq. 3</xref>. These parameters have a biological meaning, and therefore they should be fit to the corresponding measures in the neural data. However, in our case we are only aiming to fit behavioral data. As shown in <xref ref-type="fig" rid="fig7">Figure 7</xref>, and discussed in the corresponding section, <italic>&#x03C3;</italic> does not have an influence on RT, and the same mean RT can be found for different combinations of &#x03C4; and &#x0394;. In the absence of neural data, it would be meaningless to fit all parameters to RT since this would lead to overfitting. Therefore, in order to reduce the number of parameters to fit, we fix <italic>&#x03C3;&#x2009;=&#x2009;0.001&#x2009;ms<sup>&#x2212;1</sup>,</italic> and we vary &#x03C4; and &#x0394; dependently to explore the parameter space unidimensionally. Specifically, we vary <inline-formula>
<mml:math id="M60">
<mml:mrow>
<mml:mi>&#x03C4;</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mrow>
<mml:mo>[</mml:mo>
<mml:mrow>
<mml:mn>25</mml:mn>
<mml:mi mathvariant="normal">,</mml:mi>
<mml:mn>95</mml:mn>
</mml:mrow>
<mml:mo>]</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> <italic>ms</italic> and <inline-formula>
<mml:math id="M61">
<mml:mrow>
<mml:mi>&#x0394;</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>2.57</mml:mn>
<mml:mo>&#x00B7;</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mn>10</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>4</mml:mn>
</mml:mrow>
</mml:msup>
<mml:mi>&#x03C4;</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>0.0076</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> <italic>ms<sup>&#x2212;1</sup></italic>, which corresponds to the diagonal in <xref ref-type="fig" rid="fig7">Figure 7A</xref>.</p>
<p>The remaining steps of the fitting process pertain to the behavioral metrics. The second step consists of calculating the initial preferential bias <italic>&#x03D5;<sub>0</sub></italic>. Finally, in the third step, we run the model using the previously established parameters to find the best fit for <italic>&#x03C3;<sub>&#x03C8;</sub></italic> and <italic>k</italic>, i.e., the decisional uncertainty and the learning rate. Following the same argument as before, we reduced the number of parameters to fit. Since the same mean learning time can be obtained for different combinations of <italic>&#x03C3;<sub>&#x03C8;</sub></italic> and <italic>k</italic>, as shown in <xref ref-type="fig" rid="fig7">Figure 7B</xref>, we fix <italic>&#x03C3;<sub>&#x03C8;</sub></italic>&#x2009;=&#x2009;<italic>0.6</italic> and vary <inline-formula>
<mml:math id="M62">
<mml:mrow>
<mml:mi>k</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mrow>
<mml:mo>[</mml:mo>
<mml:mrow>
<mml:mn>0</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>2.5</mml:mn>
</mml:mrow>
<mml:mo>]</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
<p>To test the robustness of the fitting method, and to test whether we are overfitting, we performed a parameter recovery analysis (<xref ref-type="bibr" rid="ref92">White et al., 2018</xref>; <xref ref-type="bibr" rid="ref24">Evans et al., 2020</xref>; <xref ref-type="bibr" rid="ref17">Danwitz et al., 2022</xref>). We obtained correlation coefficients close to 1, which reflect an excellent recovery, see <xref rid="SM1" ref-type="supplementary-material">Supplementary Figure S5</xref>.</p>
<p>We fit the parameters in a sequential fashion because the estimates of both RT and VD depend uniquely on <xref ref-type="disp-formula" rid="EQ3">Eq. 3</xref>. In order to evaluate the dynamics of the perceptual processes, RT and VD are fit using horizon <italic>n<sub>H</sub> =&#x2009;0</italic> only. Once these have been established, we focus on the behavioral part, by fitting the initial preferential bias and the learning rate for different horizons.</p>
<sec id="sec11">
<label>2.5.1</label>
<title>Reaction times and visual discrimination</title>
<p>The first metric to fit is each participant&#x2019;s RT. As explained above, to perform this fit we use <xref ref-type="disp-formula" rid="EQ3">Eq. 3</xref> and data from <italic>n<sub>H</sub> =&#x2009;0</italic>, by varying <inline-formula>
<mml:math id="M63">
<mml:mrow>
<mml:mi>&#x03C4;</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mrow>
<mml:mo>[</mml:mo>
<mml:mrow>
<mml:mn>25</mml:mn>
<mml:mi mathvariant="normal">,</mml:mi>
<mml:mn>95</mml:mn>
</mml:mrow>
<mml:mo>]</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> <italic>ms.</italic> Note that due to response anticipation of the GO signal, the experimental RTs could be negative in a few cases (see <xref ref-type="fig" rid="fig3">Figure 3C</xref>). A free parameter was incorporated into the model to control for this temporal shift.</p>
<p>The second metric to fit is the VD, i.e., the ability to distinguish between stimuli. We assumed VD to be specific to each participant, and constant across blocks of each session. As a means of assessment, we checked how often the larger stimulus had been selected over the last 80 correct trials of the <italic>n<sub>H</sub> =&#x2009;0</italic> block for each level of difficulty. The only case where accuracy was low was the highest difficulty level (&#x0394;S&#x2009;=&#x2009;<italic>0.01</italic>). For our model to capture this, we used a linear transformation <inline-formula>
<mml:math id="M64">
<mml:mrow>
<mml:mover accent="true">
<mml:mi>s</mml:mi>
<mml:mo>&#x02DC;</mml:mo>
</mml:mover>
<mml:mo>=</mml:mo>
<mml:mi>&#x03B1;</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>&#x03B2;</mml:mi>
<mml:mi>s</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> to re-scale the stimuli <italic>s</italic>, ranging from 0 (empty) to 1 (full), to a range more meaningful for the model (<inline-formula>
<mml:math id="M65">
<mml:mrow>
<mml:msub>
<mml:mi>&#x03BB;</mml:mi>
<mml:mrow>
<mml:mi>L</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>R</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>~</mml:mo>
<mml:msup>
<mml:mrow>
<mml:mn>10</mml:mn>
</mml:mrow>
<mml:mrow>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msup>
</mml:mrow>
</mml:math>
</inline-formula>, (<xref ref-type="bibr" rid="ref57">Moreno-Bote et al., 2007</xref>)). Additional constraints were set for &#x03B1; and <italic>&#x03B2;</italic> so that this transformation would not swap the intensities between stimuli (i.e., if <inline-formula>
<mml:math id="M66">
<mml:mrow>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>L</mml:mi>
</mml:msub>
<mml:mo>&#x2265;</mml:mo>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>R</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> then <inline-formula>
<mml:math id="M67">
<mml:mrow>
<mml:msub>
<mml:mover accent="true">
<mml:mi>s</mml:mi>
<mml:mo>&#x02DC;</mml:mo>
</mml:mover>
<mml:mi>L</mml:mi>
</mml:msub>
<mml:mo>&#x2265;</mml:mo>
<mml:msub>
<mml:mover accent="true">
<mml:mi>s</mml:mi>
<mml:mo>&#x02DC;</mml:mo>
</mml:mover>
<mml:mi>R</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>), and so that the input stimuli would always be positive (<inline-formula>
<mml:math id="M68">
<mml:mrow>
<mml:msub>
<mml:mover accent="true">
<mml:mi>s</mml:mi>
<mml:mo>&#x02DC;</mml:mo>
</mml:mover>
<mml:mrow>
<mml:mi>L</mml:mi>
<mml:mo>,</mml:mo>
<mml:mi>R</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>&#x003E;</mml:mo>
<mml:mn>0</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>). Abiding by these conditions, we varied <italic>&#x03B1;</italic> and <italic>&#x03B2;</italic> and ran a grid-search set of simulations of <xref ref-type="disp-formula" rid="EQ3">Eq. 3</xref> (with <inline-formula>
<mml:math id="M69">
<mml:mrow>
<mml:mi>&#x0394;</mml:mi>
<mml:mi>S</mml:mi>
<mml:mo>=</mml:mo>
<mml:mrow>
<mml:mo>|</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>L</mml:mi>
</mml:msub>
<mml:mo>&#x2212;</mml:mo>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mi>R</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>|</mml:mo>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:mn>0.01</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>). We calculated how often the firing rate of the population encoding the larger stimulus was bigger than the alternative. The result depends not only on &#x03B1; and <italic>&#x03B2;</italic>, but also on <italic>&#x03C4;</italic>, <italic>&#x03C3;</italic>, and &#x0394;. Thus, to capture the large variety of results encompassed by the ranges of <italic>&#x03C4;</italic>, <italic>&#x03C3;</italic>, and &#x0394;, while abiding by the aforementioned constraints, we fix <italic>&#x03B1;</italic>&#x2009;=&#x2009;&#x2212;0.018, and let <italic>&#x03B2;</italic> vary between 0 and 0.1. These conditions allowed for proper exploration of the parameter space.</p>
<p>We ran 100-trial simulations of a horizon <italic>n<sub>H</sub> =&#x2009;0</italic> block for each combination of the parameters <italic>&#x03C4;</italic> and <italic>&#x03B2;</italic>. We then calculated the empirical cumulative distribution functions (CDF) of the RTs for all trials, and the VDs only for the difficult trials, i.e., when &#x0394;S&#x2009;=&#x2009;<italic>0.01</italic>. The distribution of simulated RTs was then compared to the distributions of experimental RTs by means of the Kolmogorov&#x2013;Smirnov distance (KSD) between CDFs (<xref ref-type="bibr" rid="ref77">Smirnov, 1948</xref>; <xref ref-type="bibr" rid="ref79">Stephens, 1974</xref>; <xref ref-type="bibr" rid="ref64">Quinn and Keough, 2002</xref>; <xref ref-type="bibr" rid="ref55">Marsaglia et al., 2003</xref>). Since both RTs and VDs strongly depend on the parameters, both were fit simultaneously. Namely, we consider the error metric <inline-formula>
<mml:math id="M70">
<mml:mrow>
<mml:mover accent="true">
<mml:mi>M</mml:mi>
<mml:mo>^</mml:mo>
</mml:mover>
<mml:mo>=</mml:mo>
<mml:mi>K</mml:mi>
<mml:mi>S</mml:mi>
<mml:mi>D</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>c</mml:mi>
<mml:mspace width="thickmathspace"/>
<mml:mo>|</mml:mo>
<mml:mi>V</mml:mi>
<mml:msup>
<mml:mi>D</mml:mi>
<mml:mrow>
<mml:mi>s</mml:mi>
<mml:mi>i</mml:mi>
<mml:mi>m</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>V</mml:mi>
<mml:msup>
<mml:mi>D</mml:mi>
<mml:mrow>
<mml:mi>r</mml:mi>
<mml:mi>e</mml:mi>
<mml:mi>a</mml:mi>
<mml:mi>l</mml:mi>
</mml:mrow>
</mml:msup>
<mml:mo>|</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>, with <italic>c</italic> being a constant set to 0.4 to balance the weight of the two metrics, and VD<sup>sim</sup>, VD<sup>real</sup> being the VD from the simulated and real data, respectively. The parameters &#x03C4; and &#x03B2; that minimize <inline-formula>
<mml:math id="M71">
<mml:mrow>
<mml:mover>
<mml:mi>M</mml:mi>
<mml:mo>&#x0302;</mml:mo>
</mml:mover>
</mml:mrow>
</mml:math>
</inline-formula> are selected for the fit. <xref ref-type="fig" rid="fig9">Figure 9A</xref> depicts the CDF of the RT for the participants and for the best-fit model simulation.</p>
<fig position="float" id="fig9">
<label>Figure 9</label>
<caption>
<p>Model fit to three sample participants&#x2019; behavioral metrics. Data used: one block of horizon 1. The specific parameter values of the fit are displayed in <xref ref-type="table" rid="tab1">Table 1</xref>. <bold>(A)</bold> Cumulative distribution function (CDF) of the reaction times (RT) for the participant data (solid red) and model simulation (dashed blue). <bold>(B)</bold> Initial bias &#x03D5;<sub>0</sub> of the participant at the beginning of the block for each trial within episode (<italic>T</italic><sub><italic>E</italic></sub>). The more the preferred choice tends toward choosing the larger (smaller) stimulus, the bigger (smaller) &#x03D5;<sub>0</sub> is. <bold>(C)</bold> Bottom: Performance of the participant (red crosses) and of the model&#x2019;s simulations (blue line: mean, shaded area: confidence interval). Top: Learning time for the participant (red dot) and model simulations (blue error bar). <bold>(D)</bold> Goodness of fit (GF) for three metrics: reaction time (RT), initial performance (PF<sub>i</sub>), and learning time (t<sub>L</sub>). Goodness of fit is calculated as follows: RT&#x2009;=&#x2009;1-Kolmogorov&#x2013;Smirnov distance between CDF, PF<sub>i</sub> =&#x2009;1- mean square error, t<sub>L</sub>: 1- difference between learning times of participant and model&#x2019;s mean divided by the total number of episodes.</p>
</caption>
<graphic xlink:href="fnbeh-18-1399394-g009.tif"/>
</fig>
<p>To summarize, in the first step of the fit, we focused on the neural dynamics layer by fitting all the free parameters of <xref ref-type="disp-formula" rid="EQ3">Eq. 3</xref>, i.e., <italic>&#x03C4;</italic> and <italic>&#x03B2;</italic>, corresponding to RT and VD. The subsequent steps consider the behavioral component of the data.</p>
</sec>
<sec id="sec12">
<label>2.5.2</label>
<title>Initial preferential bias</title>
<p>Each participant performing the task might have an initial choice preference, i.e., a natural bias toward the larger (or smaller) stimulus. In our model this is captured by the parameter <italic>&#x03D5;<sub>0</sub></italic> in <xref ref-type="disp-formula" rid="EQ4">Eq. 4</xref>. In the absence of bias, <italic>&#x03D5;<sub>0</sub></italic> equals 0.5. The greater the preference toward the bigger choice, the closer to 1 <italic>&#x03D5;<sub>0</sub></italic> will be.</p>
<p>We set a vector of initial conditions <inline-formula>
<mml:math id="M72">
<mml:mrow>
<mml:mi>&#x03D5;</mml:mi>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mi>E</mml:mi>
<mml:mo>=</mml:mo>
<mml:mn>1</mml:mn>
<mml:mi mathvariant="normal">,</mml:mi>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>E</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:msub>
<mml:mi>&#x03D5;</mml:mi>
<mml:mn>0</mml:mn>
</mml:msub>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>E</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> for each trial within episode <italic>T<sub>E</sub></italic>. To quantify <italic>&#x03D5;<sub>0</sub></italic>, we selected the first 3 episodes for each participant, and calculated the frequency <italic>f</italic> with which the larger stimulus was selected. The parameter <italic>&#x03D5;<sub>0</sub></italic> functions as an initial condition for the intended decision process (see <xref ref-type="disp-formula" rid="EQ2">Eq. 2</xref>). In agreement with the attractor dynamics, if the initial condition coincides with one of the basins of attraction, the system will be locked in that state. To prevent this (since <italic>&#x03D5;<sub>0</sub></italic> should only be an initial bias), we rescaled the frequency of the selected choices <italic>f</italic> to make the value closer to 0.5, i.e., <inline-formula>
<mml:math id="M73">
<mml:mrow>
<mml:msub>
<mml:mi>&#x03D5;</mml:mi>
<mml:mn>0</mml:mn>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>+</mml:mo>
<mml:mi>f</mml:mi>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>/</mml:mo>
<mml:mn>3</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> (other rescaling factors could be used and would not change the results). <xref ref-type="fig" rid="fig9">Figure 9B</xref> shows the values obtained for <italic>&#x03D5;<sub>0</sub></italic> for each trial within episode <italic>T<sub>E</sub></italic>. Note that we have selected one block from <italic>n<sub>H</sub> =&#x2009;2</italic> for participant 2 and <italic>n<sub>H</sub> =&#x2009;1</italic> for the others.</p>
</sec>
<sec id="sec13">
<label>2.5.3</label>
<title>Learning rate</title>
<p>Finally, to fit the remaining parameter <italic>k</italic> to each participant&#x2019;s data, we ran the model using the previously established parameters (&#x03C4;, &#x03B2;, and <italic>&#x03D5;<sub>0</sub></italic>) and fit the resulting performance to that of each participant. For each <italic>k</italic>, we ran 50 simulations and extracted the performance mean and standard deviation. To compare model and participant performances, we considered different metrics such as maximum likelihood, Bayesian (BIC) and Akaike information criterions (AIC) (<xref ref-type="bibr" rid="ref77">Smirnov, 1948</xref>; <xref ref-type="bibr" rid="ref79">Stephens, 1974</xref>; <xref ref-type="bibr" rid="ref36">Huber-Carol et al., 2002</xref>, <xref ref-type="bibr" rid="ref37">2017</xref>; <xref ref-type="bibr" rid="ref59">Nikulin and Chimitova, 2017</xref>). While these are common metrics for model comparison, they disregard the specific time dependency throughout each block, which is a key factor to characterize the learning process of the participant. Classical maximum likelihood, for example, would be strongly affected by trials exhibiting low performance due to participant fatigue or distraction. This renders the metric unsuitable for our purpose. Recently, more complex methods have been developed to overcome this issue, such as in (<xref ref-type="bibr" rid="ref8">Boelts et al., 2022</xref>). Nevertheless, we do not require such complex metrics since our goal is to show that the model can fit the full range of the participants&#x2019; data, not to compare goodness of fit to other models. To this end, we designed an <italic>ad-hoc</italic> metric consisting of two components to determine goodness of fit. The first component is the initial condition, obtained by calculating the mean-square error of the performance between the model and the data during the first five episodes. By minimizing the mean-square error, we ensured that the learning process began under similar conditions for the model and for the participant. The second factor is the time required to learn the strategy. As already outlined in the Behavioral Results section, we defined the time at which the strategy was learned as the moment after which the optimal strategy was employed in at least 9 out of the following 10 episodes, and 75% of the remaining episodes until the end of the block. To ensure that a low success rate was not due to errors caused by visual discrimination, we excluded the episodes with &#x0394;S <italic>=&#x2009;0.01</italic> from this part of the fit. In summary, by combining the results for the initial conditions (<italic>I</italic>) and the learning time (<italic>L</italic>), we could extrapolate the best fit for <italic>k</italic> by minimizing the linear combination <inline-formula>
<mml:math id="M74">
<mml:mrow>
<mml:mi>L</mml:mi>
<mml:mo>+</mml:mo>
<mml:mn>0.1.</mml:mn>
<mml:mi>I</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
<p><xref ref-type="fig" rid="fig9">Figure 9C</xref> shows the participants&#x2019; performance (red marks) as well as the associated best-fit model performance (the blue line is the mean, and the colored area is the 95% confidence interval). The top part of the plots depicts the learning time (<italic>t<sub>L</sub></italic>) calculated for the participant (red mark) as well as for the best fit model simulations (blue error-bar). <xref ref-type="table" rid="tab1">Table 1</xref> shows the best-fit parameter values per participant.</p>
<table-wrap position="float" id="tab1">
<label>Table 1</label>
<caption>
<p>Parameter values obtained when fitting data from 1 block for each of the 3 participants.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top">P.</th>
<th align="center" valign="top">
<italic>GF (RT, PF<sub>i</sub>, t<sub>L</sub>)</italic>
</th>
<th align="center" valign="top">
<italic>t<sub>L</sub></italic>
</th>
<th align="center" valign="top">
<italic>k</italic>
</th>
<th align="center" valign="top">
<italic>&#x03C4;</italic>
</th>
<th align="center" valign="top">
<italic>&#x03B2;</italic>
</th>
<th align="center" valign="top"><italic>&#x03D5;<sub>0</sub></italic> (<italic>T<sub>E</sub></italic>)</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="top">1</td>
<td align="center" valign="top">{0.91,0.96,1}</td>
<td align="center" valign="top">8</td>
<td align="center" valign="top">2.8</td>
<td align="center" valign="top">67</td>
<td align="center" valign="top">0.057</td>
<td align="center" valign="top">{0.67,0.56}</td>
</tr>
<tr>
<td align="left" valign="top">2</td>
<td align="center" valign="top">{0.87,0.62,0.97}</td>
<td align="center" valign="top">14</td>
<td align="center" valign="top">0.5</td>
<td align="center" valign="top">60</td>
<td align="center" valign="top">0.051</td>
<td align="center" valign="top">{0.56,0.67}</td>
</tr>
<tr>
<td align="left" valign="top">3</td>
<td align="center" valign="top">{0.85,0.93,1}</td>
<td align="center" valign="top">&#x2013;</td>
<td align="center" valign="top">0.4</td>
<td align="center" valign="top">67</td>
<td align="center" valign="top">0.045</td>
<td align="center" valign="top">{0.67,0.67}</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p>The parameters <italic>&#x03C4;</italic> and <italic>&#x03B2;</italic> refer to <xref ref-type="disp-formula" rid="EQ3">Eq. 3</xref>; <italic>&#x03D5;</italic><sub>0</sub> and <italic>k</italic> belong to <xref ref-type="disp-formula" rid="EQ4">Eq. 4</xref>. The learning time (t<sub>L</sub>) and the goodness of fit (GF) are shown in the first 2 columns.</p>
</table-wrap-foot>
</table-wrap>
<p>All participants except one learned the strategy yielding maximum reward value. Participant 1 learned very quickly (in just 8 episodes). The model fit to participant 1 yielded the highest learning rate (<italic>k&#x2009;=&#x2009;2.6</italic>). Interestingly, even though participant 3 did not learn the correct strategy, the parameters obtained from the fit still indicated some learning (<italic>k&#x2009;=&#x2009;0.2</italic>). Note that, though participant 2 learned the strategy fairly quickly (after only 15 episodes), participant 2&#x2019;s learning rate was only slightly greater than participant 3&#x2019;s despite participant 3 never learning the optimal strategy. The reason the learning rates for these two participants are similar, even though they reflect two distinct strategies, lies in the initial condition. Namely, participant 3 began the task with a stronger bias toward choosing the larger stimulus (<inline-formula>
<mml:math id="M75">
<mml:mrow>
<mml:msub>
<mml:mi>&#x03D5;</mml:mi>
<mml:mn>0</mml:mn>
</mml:msub>
<mml:mrow>
<mml:mo>(</mml:mo>
<mml:mrow>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>E</mml:mi>
</mml:msub>
</mml:mrow>
<mml:mo>)</mml:mo>
</mml:mrow>
<mml:mo>=</mml:mo>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:mn>0.67</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0.67</mml:mn>
</mml:mrow>
<mml:mo>}</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> vs. <inline-formula>
<mml:math id="M76">
<mml:mrow>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:mn>0.56</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0.67</mml:mn>
</mml:mrow>
<mml:mo>}</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula> for participant 2). Such disadvantageous initial conditions combined with a weak learning rate was not enough for the strategy to be learned in a block of 50 episodes.</p>
<p><xref ref-type="fig" rid="fig9">Figure 9D</xref> shows the goodness of fit for the two main behavioral metrics we aimed to reproduce: the reaction time (RT) and the performance in terms of initial performance (<italic>PF<sub>i</sub></italic>) and learning time (<italic>t<sub>L</sub></italic>). To measure the goodness of fit while remaining consistent with our fitting procedure, we used the following metrics: KSD for RT, mean-square error for <italic>PF<sub>i</sub></italic>, and the difference between the participant&#x2019;s data and the model&#x2019;s mean divided by the total number of episodes for <italic>t<sub>L</sub></italic>.</p>
<p>To summarize, we first found the best fit for the RT and VD by varying &#x03C4; and &#x03B2; in <xref ref-type="disp-formula" rid="EQ3">Eq. 3</xref>. Then, we calculated the subjective initial bias <italic>&#x03D5;<sub>0</sub></italic>. Finally, while holding the aforementioned parameters fixed, we found the best fit for the learning rate <italic>k</italic>.</p>
<p>To illustrate that the model can capture the full range of behavior, <xref ref-type="fig" rid="fig10">Figure 10</xref> shows the goodness of fit for the RT, initial performance PF<sub>i</sub>, and learning time t<sub>L</sub> for all 28 participants. For all three metrics, we show the scatter plot including each participant, the respective distribution, and the boxplot depicting the median and the 25th/75th percentiles. For reference, we superposed colored markers to indicate three sample participants shown in the previous figure.</p>
<fig position="float" id="fig10">
<label>Figure 10</label>
<caption>
<p>Goodness of fit. For RT we calculated KSD, for PF<sub>i</sub> we evaluated the mean-square error, and for t<sub>L</sub> we took the difference between the participant&#x2019;s data and the model&#x2019;s mean divided by the total number of episodes. For all three metrics, we show the scatter plot of each single participant, the corresponding distribution, and the boxplot depicting the median and the 25/75 percentiles. For reference, the superposed colored markers indicate the three participants shown in the previous figure.</p>
</caption>
<graphic xlink:href="fnbeh-18-1399394-g010.tif"/>
</fig>
<p>In summary, we fit the model to each of the participant&#x2019;s behavioral metrics. We first used the RT distribution and VD of each participant to fit the parameters in <xref ref-type="disp-formula" rid="EQ3">Eq. 3</xref>. Once these parameters were fixed, we moved on to calculate the initial bias before running simulations of the model. Finally, we compared the results of the simulations with the performance of the participants and found the best fit for the behavioral parameters, i.e., the learning rate and decisional uncertainty.</p>
</sec>
</sec>
</sec>
<sec sec-type="discussion" id="sec14">
<label>3</label>
<title>Discussion</title>
<p>In this study we analyzed how the consideration of consequence influences learning in value-based decision-making, and provided an account of the underlying neural processes. To this end, we examined how human participants learned to make sequences of decisions between value-based stimuli in the consequential task. This is a novel experimental task in which initial knowledge about environment was minimal, and explicit performance cues were absent. Consequence refers to the effect choices exert on the value of stimuli in the next trial. This was designed to promote small value choices during the early trials of each episode, and a large value one in the last trial. The instruction to each participant was to explore and to find the strategy leading to the highest cumulative reward value. The absence of explicit performance cues was meant to promote the development of a subjective assessment of performance based on relating the size of the stimuli in the current trial to the choice in the previous one. Our results show that decisions involving the computation of future consequence took longer to perform than those with no further consequence (i.e., the last choice of each episode), suggesting a more involved decision-making process when future consequence is to taken into account. Most participants eventually learned the optimal strategy, although with significant differences in their learning times.</p>
<p>Based on these observations and on previous evidence, we introduced a mathematical model of a set of plausible cognitive processes for consequence-based decision-making. The model is organized in three layers. The bottom layer describes the average dynamics of two neural populations representing the preference for each option. The populations compete against each other until their difference in activity crosses a threshold. The middle layer illustrates the participant&#x2019;s preference for choosing the bigger or smaller stimulus at each trial (the so-called intended decision). The top layer describes the strategy learning process which oversees the model&#x2019;s performance, adapts by reinforcement to maximize the cumulative reward value, and drives the intended decision layer. This oversight mechanism, combined with the modulation of preference, accurately reproduced an internal process of consequence assessment and subsequent policy update. The model was validated by fitting its parameters to reproduce each participant&#x2019;s behavioral data (i.e., reaction time distribution, visual discrimination, initial bias, and performance). The model faithfully reproduced the participants&#x2019; behavior despite its varied nature. Importantly, this model also provides a plausible account of the neural processes required for gauging options as a function of their associated consequence (measured in terms of reward), and of how these processes are involved in decision-making.</p>
<sec id="sec15">
<label>3.1</label>
<title>Justification of the consequential task</title>
<p>Real world decisions are rarely accompanied by immediate feedback, there is often a conflict between short and long-term reward, consequences are often long-lasting, reward is often difficult to quantify, and state-action spaces often require exploration to define (as opposed to being known <italic>a priori</italic>). Several of these characteristics generate uncertainty and complicate performance assessment. The consequential task combined features common to both hierarchical decision-making (<xref ref-type="bibr" rid="ref52">Lorteije et al., 2015</xref>; <xref ref-type="bibr" rid="ref98">Zylberberg et al., 2017</xref>; <xref ref-type="bibr" rid="ref97">Zylberberg, 2022</xref>) and delay discounting paradigms (<xref ref-type="bibr" rid="ref32">Hayden and Platt, 2007</xref>; <xref ref-type="bibr" rid="ref43">Kim et al., 2008</xref>; <xref ref-type="bibr" rid="ref38">Hwang et al., 2009</xref>; <xref ref-type="bibr" rid="ref1">Alexander and Brown, 2010</xref>; <xref ref-type="bibr" rid="ref31">Hayden, 2016</xref>) to examine how this kind of decision-making unfolds. Moreover, the absence of cued performance feedback during the task made our paradigm particularly suitable for studying how learning optimal strategies may extend from immediate perceptual decision-making to a more complex process involving predictions of future states. Unlike standard hierarchical decision-making and partially observable Markov decision processes (<xref ref-type="bibr" rid="ref76">Smallwood and Sondik, 1973</xref>; <xref ref-type="bibr" rid="ref40">Kaelbling et al., 1998</xref>), participants in the consequential task were not aware of the underlying relationship between actions and their consequences. Participants were told only that the choice they made in one trial might influence the next. In this way, participants had to explore and observe the consequences of their choices to deduce that an inter-trial dependence existed. Moreover, participants could never be certain if they found the optimal solution, i.e., picked the correct sequence of decisions to maximize cumulative reward value. This is in sharp contrast to delay discounting tasks which largely focus on the principle of inhibitory short-term control where the presence of explicit cues helps overcome impulsive behavior, such as in the <italic>farming on Mars</italic> task (<xref ref-type="bibr" rid="ref30">Gureckis and Love, 2009</xref>) (see below).</p>
</sec>
<sec id="sec16">
<label>3.2</label>
<title>Cognitive hypothesis based on behavioral results</title>
<p>The purpose of our study was to understand how participants learned how their choices influenced the decision context, as opposed to assessing whether reward value varied with time. In other words, the absence of explicit cues was intended to force the participants to rely on their own subjective assessment to infer the delayed consequence of their decisions across groups of successive trials, and whether their strategy was being successful. This inner assessment had to be driven by the participant&#x2019;s probing of patterns of action/decision effects. Complementary to this, we believe that participants had to go through a hypothesis testing process, until the eureka moment of realizing that one specific strategy was better than the others. Consequently, to find the optimal strategy, participants had to first realize that choosing the smaller option lead to more rewarding options (the eureka moment). Explicitly, this implies identifying the specific feature of the stimuli to be considered, having nothing else than the observance of their choice/action effects on the environment (the stimuli in the next trial). Then, they had to confirm their criterion based on the global effect of their choices on the stimuli size across episodes.</p>
</sec>
<sec id="sec17">
<label>3.3</label>
<title>Rule-based vs. Far-sighted assessment of consequence</title>
<p>The strategy to attain the highest possible cumulative reward value may be operationalized as a sequence of decision rules: choose small, then big in horizon 1 episodes; choose small, then small, then big, in horizon 2 episodes. Though we expected the participants&#x2019; choices to abide by these rules once the learning was complete and the optimal decision strategy was established, the focus of this study is on how consequence-based assessment forms and influences the learning of that optimal strategy. Because of this, it was crucial that the consequential task were devoid of any cued performance feedback, which could potentially inform the participant of his/her performance after each episode and, ultimately, promote a rule-based strategy.</p>
<p>For the same purpose, and to promote exploration, the participants were left with the uncertainty of neither having a criterion to follow to make decisions nor the knowledge about which aspect of the stimuli to attend to while making decisions. Note that, in addition to the bar heights (proportional to reward value), the stimuli at each trial were presented on the right and left of the screen, they were shown sequentially, randomly alternating their order of presentation across trials. Both the position and order of presentation of the stimuli increased the uncertainty with respect to the relevant stimuli dimensions. Under these conditions, participants had to perceive the relationship between their choices and the values of the stimuli presented in subsequent trials. If noticed, this observation could then be used to predict the consequence associated with choosing each option at each trial within episode. In other words, participants had to identify the relevant aspects of the stimuli for the goal at hand and rely on their own subjective perception of performance. This derived from their observations of the stimuli presented after each decision and by their own internal assessment criterion which itself was based on their ability to estimate the sum of water (reward value) throughout the trials of each episode.</p>
<p>To summarize, cued performance feedback could have reduced task to simple rule-based learning. Although the optimal strategy consists of a rule-based sequence, the crucial element of the task is that the participant must undergo a phase of exploration in which learning is driven by exploration and assessment of the reward-based consequence associated with each option.</p>
</sec>
<sec id="sec18">
<label>3.4</label>
<title>Computational model of consequence</title>
<p>Drift-diffusion models (DDM) have been used to describe how sensory decisions unfold as a function of evidence accumulation (<xref ref-type="bibr" rid="ref65">Ratcliff and McKoon, 2008</xref>). Likewise, urgency-gating model (<xref ref-type="bibr" rid="ref15">Cisek et al., 2009</xref>) emphasize the contribution of the passage of time to make sensory based decisions in dynamic environments. Extended versions of the DDM have also been used to describe how evidence relates to value-based decisions via informative saccades (<xref ref-type="bibr" rid="ref46">Krajbich et al., 2010</xref>; <xref ref-type="bibr" rid="ref47">Krajbich and Rangel, 2011</xref>), extending into hybrid models that can adapt their parameters over time (<xref ref-type="bibr" rid="ref25">Fontanesi et al., 2019</xref>; <xref ref-type="bibr" rid="ref8">Boelts et al., 2022</xref>) via reinforcement learning (<xref ref-type="bibr" rid="ref80">Sutton and Barto, 1981</xref>). However, these formulations fall short to describe the complexity of brain population dynamics during decision-making and of inhibitory processes therein. Furthermore, they do not capture how action effects and rewards are subjectively perceived and merged in contexts in which these are delayed and must be first perceived and learned, as it occurs in the consequential task. In brief, here we intended a formalization of the neural processes underlying reward-driven, delayed-value, multi-step decisions in a context in which attaining reward is contingent on learning the covert effect of actions on the environment. In this way, learning must operate in the absence of explicit performance feedback, and in the absence of knowledge of the target strategy itself, which is unlike previous RL-based formulations. By contrast, if the purpose of the present study were to merely provide an estimate of the participants&#x2019; decisions and learning process, an RL formulation could have been used to solve the credit assignment problem (<xref ref-type="bibr" rid="ref56">Minsky, 1961</xref>) and learn the behavioral strategy. However, these models fall short of the aforementioned aspects of neuronal dynamics, competition and inhibition that we targeted in this study.</p>
<p>Learning in our model is operationalized by a reinforcement comparison algorithm (<xref ref-type="bibr" rid="ref2">Amari, 1998</xref>; <xref ref-type="bibr" rid="ref10">Brunel and Wang, 2001</xref>; <xref ref-type="bibr" rid="ref67">Roxin and Ledberg, 2008</xref>; <xref ref-type="bibr" rid="ref46">Krajbich et al., 2010</xref>; <xref ref-type="bibr" rid="ref16">Cos et al., 2013</xref>; <xref ref-type="bibr" rid="ref73">Shahar et al., 2019</xref>), scaled by the difference between predicted vs. obtained reward value (<xref ref-type="bibr" rid="ref80">Sutton and Barto, 1981</xref>; <xref ref-type="bibr" rid="ref18">Dayan, 1992</xref>), measured accordingly to the participant&#x2019;s subjectively perceived scale. For simplicity, we assumed a fixed function across participants to quantify reward value [R(T) function in <xref ref-type="disp-formula" rid="EQ4">Eq. 4</xref>]. Furthermore, to provide the necessary flexibility for the model to capture the full range of participants&#x2019; learning dynamics, the model included two free parameters, the learning rate and the decisional uncertainty, to be fit to each participant&#x2019;s behavior. The result is a model that could faithfully reproduce the full range of behaviors of each participant: RT distribution, pattern of decision-making, and learning time.</p>
<p>The model is organized in three layers. The lower neural dynamics layer represents the average activity of two neural populations competing for selection, each sensitive to one of the two stimuli at each trial. The commitment for an option is made when the difference in firing rate between the two populations crosses a given threshold (<xref ref-type="bibr" rid="ref2">Amari, 1998</xref>; <xref ref-type="bibr" rid="ref10">Brunel and Wang, 2001</xref>; <xref ref-type="bibr" rid="ref54">Marcos et al., 2013</xref>). A similar architecture, with small variations, has been used to model decision-making in a broad set of tasks (<xref ref-type="bibr" rid="ref96">Wong and Wang, 2006</xref>; <xref ref-type="bibr" rid="ref54">Marcos et al., 2013</xref>; <xref ref-type="bibr" rid="ref53">Marcos and Genovesio, 2016</xref>; <xref ref-type="bibr" rid="ref50">Lam et al., 2022</xref>) and can describe most types of single-trial, binary decision-making, including value-based and perceptual paradigms. Importantly, our model does not provide a clear delineation between deliberation and commitment as DDMs do, but rather a neuron-like unselective ramp-up representation of options that diverge until a commitment is made. Like accumulation-to-bound models, attractor-based models can also account for speed-accuracy trade-offs during decision-making. We chose this kind of formalism because attractor models are more biologically realistic than the abstract accumulation-to-bound ones, and possibly provide a more promising avenue for unifying theories of brain and behavior. This was necessary for our model to provide a plausible explanation for the neural competition and inhibition known to operate in premotor and prefrontal cortical areas. Moreover, our model weighs inputs with recurrent activity during sequences of decisions and projects this formulation for a neighbor neurophysiological study. Note that this layer of the model can be derived analytically from a network of spiking neurons used for making binary decisions (<xref ref-type="bibr" rid="ref88">Wang, 2002</xref>). Beyond the scope of this study, this model could also subserve probing into working memory (<xref ref-type="bibr" rid="ref19">Deco and Rolls, 2005</xref>; <xref ref-type="bibr" rid="ref96">Wong and Wang, 2006</xref>); a transient input could bring the system from the resting state to one of the two stimulus-selective persistent activity states, to be internally maintained across a delay period.</p>
<p>In addition to binary population competition, we claim that modeling consequence-based decision-making requires at least two additional mechanisms. The first one is needed to prioritize a specific policy to guide the decisions; the second one to create an internal mechanism of performance to evaluate these criteria, based on the difference between predicted and obtained reward value. Accordingly, the role of the middle layer (intended decision) is to implement those criteria which in our case depend on the relative value of the stimuli and on the number of trial within episode. Finally, the top layer (strategy learning) carries out learning by reinforcement comparison (<xref ref-type="bibr" rid="ref81">Sutton and Barto, 2018</xref>) and temporal difference (<xref ref-type="bibr" rid="ref80">Sutton and Barto, 1981</xref>; <xref ref-type="bibr" rid="ref35">Houk et al., 1995</xref>).</p>
<p>Altogether, our model introduces a plausible implementation of the neurocognitive processes involved in consequence-based decision-making. Each part of the model is essential to describe decision-making, inhibition, and learning. For the neural dynamics layer, the set of equations corresponds to the most simplified version of a network of brain neurons during binary decision-making (<xref ref-type="bibr" rid="ref96">Wong and Wang, 2006</xref>); it makes use of only two populations of neurons and a minimal set of parameters. The middle layer consists of one equation (with only one free parameter) and makes use of the simplest possible form of a two-attractor dynamical system (with the addition of a noise component). Finally, the top layer follows a reinforcement comparison algorithm, and adds a single free parameter to the model: the learning rate. Each of these elements is indispensable for a biologically plausible theoretical formalization of consequence-based decision-making. Without the first layer we would not have a biologically plausible decision-making model, without the middle layer we could not describe policy changes, and without the top layer we would not have learning.</p>
<p>Previous research describes models of learning processes during decision-making, for the most part implemented via RL (<xref ref-type="bibr" rid="ref81">Sutton and Barto, 2018</xref>). Although our paradigm could also be modeled with RL, the clear advantage of our model is that it does not only describe the behavioral patterns of learning for each individual participant, but provides a biophysically plausible description of the underlying brain processes when predicting RTs. Moreover, our model is directly grounded on the neural substrate dynamics, since the mean-field approximation has been derived analytically from networks of spiking neurons (<xref ref-type="bibr" rid="ref88">Wang, 2002</xref>).</p>
<p>The results and predictions depicted in the model show that the dynamics of the three layers combined can accurately reproduce the behavior of each single participant, including those who did not attain the optimal strategy. The low number (4) of equations in the model, together with the low number of free parameters (7, of which only 3 are used for fitting), makes this model a simple, yet powerful tool able to reproduce a large variety of behaviors. Moreover, unlike the basic RL agents or models for evidence accumulation, our model is biologically plausible and predicting individual behavioral metrics, such as RT, initial bias, and visual discrimination. Note that, for the behavioral part of the model, only one free parameter is used, i.e., learning rate. A larger number of free parameters (at least 3) is needed for classical reinforcement learning algorithms, e.g., Q-learning.</p>
<p>The comprehensive formulation of the model makes it possible to explain and fit various scenarios. We have already mentioned the differences in learning speeds, and that the model could fit any of them, even when there was no learning. Another example is the difference in the order of execution of the blocks. Namely, most participant were able to take the optimal strategy learned in one horizon and generalize it to the other horizon block, making the learning much faster (see <xref rid="SM1" ref-type="supplementary-material">Supplementary materials</xref>). In our model, this is captured mainly by the initial bias which is calculated for each block individually. As third example, potentially, a characteristic that our model could fit is the difference in RT between trials within episodes and horizons (see <xref ref-type="fig" rid="fig2">Figure 2F</xref>). In this manuscript, for simplicity, we decided to perform a single fit for the neural dynamics&#x2019; equations, finding one set of parameters per participants. To explain the differences between horizons and trials within episodes, the same fit should be done for each condition. Moreover, even if it is not the case of this specific task, the model is able to adapt in case of a sudden change of strategy. Nevertheless, if this would be the case, it would be advisable to adopt a more realistic adaptation mechanism. Namely, it seems reasonable to assume that, after learning, once a participant realizes that the optimal strategy used so far is not working anymore, he would reset his strategy instead of gradually change it. However, even though it is an interesting topic, this is work for future investigation.</p>
</sec>
</sec>
<sec id="sec19">
<label>4</label>
<title>Conclusion and future work</title>
<p>In this manuscript we have introduced a minimalistic formalism of the brain dynamics of consequence-based decision-making and its associated learning process. We validated this formalism with the behavioral data gathered from 28 human participants, which the model could accurately reproduce. By extending classic, single-trial binary decision-making, we designed a oversight mechanism based on the assessment of the effect of decisions on subsequent stimuli, and a reinforcement rule to modify behavioral preferences. We also designed the consequential task, an experimental framework in which acquiring the most reward value required learning to assess the consequence associated with each option during the decision-making process. Both the experimental results and the model predictions describe consequence-based decision-making as an extended version of value-based decision-making in which the computation of predicted reward value may extend over several trials. The formalism introduces the necessary notions of oversight of the current strategy and of adaptive reinforcement, as the minimal requirements to learn consequence-based decision-making.</p>
<p>Although our model has been designed and tested in the consequential task described here, we argue that its generalization to similar paradigms in which optimal decisions require assessing the consequence associated to the presented options, or sequences of multiple decisions, may be relatively straightforward. Specifically, we envision three possible future extensions to facilitate its generalization. First, the model could incorporate several preference criteria (either simultaneously or combinations thereof) into the intended decision layer: left vs. right or first vs. second, instead of small vs. big, to be determined in a dynamical fashion. This could be achieved with a multi-dimensional attractor model, with as many basins of attraction as the number of preference criteria to be considered.</p>
<p>The second future extension is the re-definition of the reward function R(T) according to the subjective criterion of preference. Namely, a reward value can be perceived differently by different participants, i.e., people operate optimally according to their own subjective perception of the reward value. Because of this, a possible extension is to incorporate an individual reward value function per participant (R(T) in <xref ref-type="disp-formula" rid="EQ4">Eq. 4</xref>). For simplicity, in this manuscript we set R(T) to be fixed and to be the objective reward value function. In case a participant did not perceive what was the optimal reward value, he/she performed sub-optimally according to objective reward function, and the model responded by allowing the learning constant <italic>k</italic> to be zero. This holds since the optimal strategy was never reached, and the fitting of the participant&#x2019;s performance was correct. Nevertheless, it remains a standing work of significant interest to investigate different subjective reward mechanisms and their implementation in the model.</p>
<p>Finally, the third enhancement we propose for our model is making the learning rate time dependent, i.e., <italic>k</italic>(<italic>E</italic>). This would facilitate reproducing learning processes starting at different times throughout the session. For example, it is possible that participants initiate the session having in mind a possible (incorrect) strategy and they stick to it without looking for clues, and therefore without learning the optimal policy. Nevertheless, after many trials they may change their mind and begin to explore different strategies. In this case, the learning rate <italic>k</italic>(<italic>E</italic>) would be set to zero for all the initial trials when indeed there is no learning.</p>
<p>Again, we want to emphasize that even if this model is built for the consequential task, it contains all the elements and processes to reproduce behavior from other tasks involving sequential consequence-based decision-making. Note that the strategy learning mechanism is already general enough to adapt to tasks where the optimal policy is not fixed throughout the experiment. In the case of a policy reversal, for example, the learning mechanism would be able to detect a change and adapt accordingly. Finally, we want to stress that our model could be applied to other decision-making paradigms, such as a version of the consequential random-dot task (<xref ref-type="bibr" rid="ref9">Britten et al., 1993</xref>) or other multiple-option paradigms.</p>
</sec>
<sec sec-type="materials|methods" id="sec20">
<label>5</label>
<title>Materials and methods</title>
<sec id="sec21">
<label>5.1</label>
<title>Participants</title>
<p>A total of 28 participants (15 males, 13 females; age range 18&#x2013;30&#x2009;years; all right hand dominant) participated in the experimental task. All participants were neurologically healthy, had normal or corrected to normal vision, were naive as to the purpose of the study, and gave informed consent before participating. The study was approved by the local Clinical Research Ethics Committee (CEIm Ref. #2021/9743/I) and was conducted in accordance with relevant guidelines and regulations. Participants were paid a &#x20AC;10 show-up fee.</p>
</sec>
<sec id="sec22">
<label>5.2</label>
<title>Experimental setup</title>
<p>Participants were situated in the laboratory room at the Facultat de Matem&#x00E0;tiques i Inform&#x00E0;tica, Universitat de Barcelona, where the task was performed. The participants were seated in a chair, facing the experimental table, with their chest approximately 10&#x2009;cm from the table edge and their right arm resting on its surface. The table defined the plane where reaching movements were to be performed by sliding a light computer mouse (Logitech Inc). On the table, approximately 60&#x2009;cm away from the participant&#x2019;s sitting position, we placed a vertically-oriented, 24&#x201D; Acer G245HQ computer screen (1920&#x00D7;1080). This monitor was connected to an Intel i5 (3.20GHz, 64-bit OS, 8&#x2009;GB RAM) portable computer that ran custom-made scripts, programmed in MATLAB with the help of the MonkeyLogic toolbox, to control task flow (NIMH MonkeyLogic, NIH, United States; <ext-link xlink:href="https://monkeylogic.nimh.nih.gov/" ext-link-type="uri">https://monkeylogic.nimh.nih.gov</ext-link>). The screen was used to show the stimuli at each trial and the position of the mouse in real time.</p>
<p>As part of the experiment, the participants had to respond by performing overt movements with their arm along the table plane while holding the computer mouse. Their movements were recorded with a Mouse (Logitech, Inc), sampled at 1&#x2009;kHz, which we used to track hand position. Given that the monitor was placed upright on the table and movements were performed on the table plane (horizontally, approximately from the center of the table to the left or right target side), the plane of movement was perpendicular to that of the screen, where the stimuli and finger trajectories were presented. Data analyses were performed with custom-built MATLAB scripts (The Mathworks, Natick, MA), licensed to the Universitat de Barcelona.</p>
</sec>
<sec id="sec23">
<label>5.3</label>
<title>Consequential decision-making task</title>
<p>This section describes the consequential decision-making task, designed to assess the role of consequence on decision-making while promoting prefrontal inhibitory control (<xref ref-type="bibr" rid="ref91">Wessel and Aron, 2017</xref>). Since consequence depends on a predictive evaluation of future contexts, we designed a task in which trials were grouped together into episodes (groups of one, two or three consecutive trials), establishing the horizon of consequence for the decision-making problem within that block of trials.</p>
<p>The number of trials per episode equals the horizon <italic>n<sub>H</sub></italic> plus 1. In brief, within an episode, a decision in the initial trial influences the stimuli to be shown in the next trial(s) in a specific fashion, unbeknown to our participants. Although a reward value is gained by selecting one of the stimuli presented in each trial, the goal is not to gain the largest amount as possible per trial, but rather per episode.</p>
<p>Each participant performed 100 episodes for each horizon <italic>n<sub>H</sub></italic>&#x2009;=&#x2009;0, 1, and 2. In the interest of comparing results, we have generated a list of stimuli for each <italic>n<sub>H</sub></italic> and used it for all participants. To avoid fatigue and keep the participants focused, we divided the experiment into 6 blocks, to be performed on the same day, each consisting of approximately 100 trials. More specifically, there was 1 block of <italic>n<sub>H</sub> =&#x2009;0</italic> with 100 trials, 2 blocks of <italic>n<sub>H</sub> =&#x2009;1</italic> each with 100 trials, and 3 blocks of <italic>n<sub>H</sub> =&#x2009;2</italic> with two of them of 105 trials and one of 90. Finally, we have randomized the order in which participants performed the horizons.</p>
<p><xref ref-type="fig" rid="fig1">Figure 1</xref> shows the timeline of one <italic>n<sub>H</sub></italic>&#x2009;=&#x2009;1 episode (2 consecutive trials). The episode consists of two dependent trials. At the beginning of the trial, the participant was required to move the cursor onto a central target. After a fixation time (500&#x2009;ms), the two target boxes were shown one after the other (for 500&#x2009;ms each) to the left and right of the screen, in a random order. Targets were rectangles filled in blue by a percentage corresponding to the reward value associated with each stimulus (analogous to water containers). Next, both targets were presented together. This served as the GO signal for the participant to choose one of them (within an interval of 4&#x2009;s). Participants had to report their choice by making a reaching movement with the computer mouse from the central target to the target of their choice (right or left container). If the participant did not make a choice within 4&#x2009;s, the trial was marked as an error trial. Once one of the targets had been reached for and the participant had held that position (500&#x2009;ms), the selection was recorded, and a yellow dot appeared above the selected target, indicating successful selection and reward value acquisition. In case of horizons larger than 0, the second trial started following the same pattern, although with a set of stimuli that depended on the previous decision (see next section). A progress bar at the bottom of the screen indicates the current trial within the episode (for <italic>n<sub>H</sub></italic>&#x2009;=&#x2009;1, 50% during the first trial, 100% during the second trial).</p>
<p>At the beginning of the session, participants were given instructions on how to perform the task. Specifically, using some sample trials, we demonstrated them how to select a stimulus by moving the mouse. Step by step we showed that a target appears in the center of the screen indicating the start of an episode. We told them that they had 4&#x2009;s to move the cursor to the central cross. After moving the cursor to the central cross, two bars appear, one after the other, and once both appear together/simultaneously, they had 4&#x2009;s to make their decision by moving the cursor over one of the two bars. At that point a yellow dot appears over the bar indicating their selection. After that, the central target appears again indicating the beginning of a new trial. After explaining how to technically execute the task, we focused on explaining the task goal. We showed them a schematic of the task, much like the one in <xref ref-type="fig" rid="fig1">Figure 1A</xref> illustrating the structure of trials and episodes. We told them that the goal is to get as much reward (water) as possible in each episode, and that for episodes with more than 1 trial each, the choice in a trial may have an effect on what appears in the next trial in the same episode. We encouraged them to explore in order to try to figure out what that effect might be, while keeping in mind that their goal is always to maximize the total reward in each episode. Finally, we told them that they will be presented with a series of episodes in a row, each episode is independent, meaning that their decisions in one episode have no effect on subsequent ones.</p>
</sec>
<sec id="sec24">
<label>5.4</label>
<title>Episode structure</title>
<p>The participants were instructed to maximize the cumulative reward value throughout each episode, namely the sum of water contained by the selected targets across the trials of the episode. If trials within an episode were independent, the optimal choice would be to always choose the largest stimulus. Since one of the major goals of our study was to investigate delayed consequence assessment involving adaptive choices, we deliberately created dependent trial contexts in which making incentive decisions (selecting the larger stimulus) would not lead to the most cumulative reward value within episode.</p>
<p>To promote inhibitory choices, the inter-trial relationship was designed such that selecting the small (large) stimulus in a trial, yielded an increase (decrease) in the mean value of the options presented in the next trial. As explained below, because of the parameters choice we made, always choosing the larger stimulus did not maximize cumulative reward value for <italic>n<sub>H</sub> =&#x2009;1, 2.</italic></p>
<p>Trials were generated according to 3 parameters: horizon&#x2019;s depth <italic>n<sub>H</sub></italic>, perceptual discrimination (in terms of difference <inline-formula>
<mml:math id="M77">
<mml:mrow>
<mml:mi>&#x0394;</mml:mi>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula> between the stimuli), and the gain/loss <italic>G</italic> in mean size of stimuli for successive trials. The stimuli <inline-formula>
<mml:math id="M78">
<mml:mrow>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> presented on the screen could take values ranging from 0 to 1. Trials were divided into five difficulty levels by setting the difference between stimuli <inline-formula>
<mml:math id="M79">
<mml:mrow>
<mml:mi>&#x0394;</mml:mi>
<mml:mi>S</mml:mi>
<mml:mo>&#x2208;</mml:mo>
<mml:mrow>
<mml:mo>{</mml:mo>
<mml:mrow>
<mml:mn>0.01</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0.05</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0.1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0.15</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>0.2</mml:mn>
</mml:mrow>
<mml:mo>}</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
<p>For horizon <italic>n<sub>H</sub> =&#x2009;0</italic>, for each trial the stimuli <inline-formula>
<mml:math id="M80">
<mml:mrow>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> are generated as to have mean <italic>M</italic> and difference <italic>d</italic> between them, i.e., <inline-formula>
<mml:math id="M81">
<mml:mrow>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mi>M</mml:mi>
<mml:mo>&#x00B1;</mml:mo>
<mml:mi>&#x0394;</mml:mi>
<mml:mi>S</mml:mi>
<mml:mo>/</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>. To have stimuli ranging from 0 to 1, the mean <italic>M</italic> is randomly generated using a uniform distribution with bounds <inline-formula>
<mml:math id="M82">
<mml:mrow>
<mml:mrow>
<mml:mo>[</mml:mo>
<mml:mrow>
<mml:mi>&#x0394;</mml:mi>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mrow>
<mml:mi>max</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>/</mml:mo>
<mml:mn>2</mml:mn>
<mml:mi mathvariant="normal">,</mml:mi>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x0394;</mml:mi>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mrow>
<mml:mi>max</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>/</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
<mml:mo>]</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>, where <inline-formula>
<mml:math id="M83">
<mml:mrow>
<mml:mi>&#x0394;</mml:mi>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mrow>
<mml:mi>max</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mn>0.2</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> is the maximum &#x0394;S. In horizon <italic>n<sub>H</sub> =&#x2009;1</italic>, each episode consists of 2 dependent trials. Specifically, the stimuli presented in the second trial depend on the selection reported in the previous trial of that same episode. More specifically, the rule is such that if the choice of the first trial is the smaller/larger stimulus, the mean of the pair of stimuli in the second trial will be increased/decreased by a specific gain <italic>G</italic>. In practice, the first trial of an <italic>n<sub>H</sub> =&#x2009;1</italic> episode is generated in the same way as for horizon <italic>n<sub>H</sub> =&#x2009;0</italic>, i.e., the two stimuli equal <inline-formula>
<mml:math id="M84">
<mml:mrow>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mi>M</mml:mi>
<mml:mo>&#x00B1;</mml:mo>
<mml:mi>&#x0394;</mml:mi>
<mml:mi>S</mml:mi>
<mml:mo>/</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>. The stimuli in the second trial within the same episode could be either <inline-formula>
<mml:math id="M85">
<mml:mrow>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mi>M</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>G</mml:mi>
<mml:mo>&#x00B1;</mml:mo>
<mml:mi>&#x0394;</mml:mi>
<mml:mi>S</mml:mi>
<mml:mo>/</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula> or <inline-formula>
<mml:math id="M86">
<mml:mrow>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mi>M</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>G</mml:mi>
<mml:mo>&#x00B1;</mml:mo>
<mml:mi>&#x0394;</mml:mi>
<mml:mi>S</mml:mi>
<mml:mo>/</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>, depending on the previous decision. Note that the difficulty of the trial remains constant within episode. A schematic for the trial structure is shown in <xref ref-type="fig" rid="fig1">Figure 1</xref>. Again, to have stimuli ranging from 0 to 1, the mean <italic>M</italic> is randomly generated using a uniform distribution with bounds <inline-formula>
<mml:math id="M87">
<mml:mrow>
<mml:mrow>
<mml:mo>[</mml:mo>
<mml:mrow>
<mml:mi>G</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>&#x0394;</mml:mi>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mrow>
<mml:mi>max</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>/</mml:mo>
<mml:mn>2</mml:mn>
<mml:mi mathvariant="normal">,</mml:mi>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>G</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x0394;</mml:mi>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mrow>
<mml:mi>max</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>/</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
<mml:mo>]</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>. In horizon <italic>n<sub>H</sub> =&#x2009;2</italic>, episodes consist of three trials. The trial generation is structured as for horizon <italic>n<sub>H</sub> =&#x2009;1</italic>. Namely, the first trial has stimuli <inline-formula>
<mml:math id="M88">
<mml:mrow>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mi>M</mml:mi>
<mml:mo>&#x00B1;</mml:mo>
<mml:mi>&#x0394;</mml:mi>
<mml:mi>S</mml:mi>
<mml:mo>/</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>, the second <inline-formula>
<mml:math id="M89">
<mml:mrow>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mi>M</mml:mi>
<mml:mo>&#x00B1;</mml:mo>
<mml:mi>G</mml:mi>
<mml:mo>&#x00B1;</mml:mo>
<mml:mi>&#x0394;</mml:mi>
<mml:mi>S</mml:mi>
<mml:mo>/</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>, and the third <inline-formula>
<mml:math id="M90">
<mml:mrow>
<mml:msub>
<mml:mi>s</mml:mi>
<mml:mrow>
<mml:mn>1</mml:mn>
<mml:mo>,</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:msub>
<mml:mo>=</mml:mo>
<mml:mi>M</mml:mi>
<mml:mo>&#x00B1;</mml:mo>
<mml:mi>G</mml:mi>
<mml:mo>&#x00B1;</mml:mo>
<mml:mi>G</mml:mi>
<mml:mo>&#x00B1;</mml:mo>
<mml:mi>&#x0394;</mml:mi>
<mml:mi>S</mml:mi>
<mml:mo>/</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
</mml:math>
</inline-formula>. To have stimuli ranging from 0 to 1, the mean <italic>M</italic> is randomly generated from a uniform distribution with bounds <inline-formula>
<mml:math id="M91">
<mml:mrow>
<mml:mrow>
<mml:mo>[</mml:mo>
<mml:mrow>
<mml:mn>2</mml:mn>
<mml:mi>G</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>&#x0394;</mml:mi>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mrow>
<mml:mi>max</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>/</mml:mo>
<mml:mn>2</mml:mn>
<mml:mi mathvariant="normal">,</mml:mi>
<mml:mn>1</mml:mn>
<mml:mo>&#x2212;</mml:mo>
<mml:mn>2</mml:mn>
<mml:mi>G</mml:mi>
<mml:mo>&#x2212;</mml:mo>
<mml:mi>&#x0394;</mml:mi>
<mml:msub>
<mml:mi>S</mml:mi>
<mml:mrow>
<mml:mi>max</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>/</mml:mo>
<mml:mn>2</mml:mn>
</mml:mrow>
<mml:mo>]</mml:mo>
</mml:mrow>
</mml:mrow>
</mml:math>
</inline-formula>. We set the gain/loss parameter to <italic>G&#x2009;=&#x2009;0.3</italic> and <italic>G&#x2009;=&#x2009;0.19</italic> for horizon <italic>n<sub>H</sub> =&#x2009;1</italic> and <italic>n<sub>H</sub></italic>&#x2009;=&#x2009;2, respectively. Our choice was motivated by the fact that G should be big enough to have a deterministic optimal strategy, i.e., always choosing the smaller reward value apart from the last trial within episode. In other words, choosing the bigger stimulus never compensates for the loss given by G. Moreover, <italic>G</italic> should be big enough to let the participants perceive the gain/loss between trials, while simultaneously allowing some variability for the randomly generated means <italic>M</italic>.</p>
</sec>
<sec id="sec25">
<label>5.5</label>
<title>Statistical analysis</title>
<p>The dependency of PF and RT on VD together with the other variables must be established statistically. To assess the learning process, we quantified the relationship of PF and RT with horizon <italic>n<sub>H</sub></italic>, trial within episode <italic>T<sub>E</sub></italic>, and episode <italic>E</italic>. To obtain consistent results, we adjusted these variables as follows. In the calculation, the trial within episode is reversed, from last to first, because the optimal choice for the last <italic>T<sub>E</sub></italic> (large) is the same regardless of the horizon number. Furthermore, regarding the model for PF, to consider trials within episode independently, we adapted the notion of PF (defined as a summary measure per episode) to an equivalent of PF per trial, i.e., the probability of choosing the optimal choice <inline-formula>
<mml:math id="M92">
<mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi>o</mml:mi>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>. Finally, to assess the difference between learning groups, we introduce the categorical variable L that identifies the group of participants that learned the optimal strategy and the ones who did not, according to <xref ref-type="fig" rid="fig2">Figure 2A</xref>. We then used a generalized linear mixed effects model (<xref ref-type="bibr" rid="ref85">Verbeke and Molenberghs, 2009</xref>; <xref ref-type="bibr" rid="ref26">Ga&#x0142;ecki and Burzykowski, 2013</xref>) to predict PF and RT. The independent variables for the fixed effects are horizon <italic>n<sub>H</sub></italic>, trial within episode <inline-formula>
<mml:math id="M93">
<mml:mrow>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>E</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, and the passage of time expressed in terms of episodes <italic>E,</italic> and &#x0394;S. We set the random effects for the intercept and the episodes grouped by participant <italic>p</italic>; we write the random effects as <inline-formula>
<mml:math id="M94">
<mml:mrow>
<mml:mi mathvariant="normal">(</mml:mi>
<mml:mi>E</mml:mi>
<mml:mi mathvariant="normal">|</mml:mi>
<mml:mi>p</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>. The resulting models are: <inline-formula>
<mml:math id="M95">
<mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi>o</mml:mi>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>~</mml:mo>
<mml:mi>L</mml:mi>
<mml:mo>&#x00B7;</mml:mo>
<mml:mi>E</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>L</mml:mi>
<mml:mo>&#x00B7;</mml:mo>
<mml:mi>&#x0394;</mml:mi>
<mml:mi>S</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>L</mml:mi>
<mml:mo>&#x00B7;</mml:mo>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mi>H</mml:mi>
</mml:msub>
<mml:mo>&#x00B7;</mml:mo>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>E</mml:mi>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:mi mathvariant="normal">(</mml:mi>
<mml:mi>E</mml:mi>
<mml:mi mathvariant="normal">|</mml:mi>
<mml:mi>p</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula> and <inline-formula>
<mml:math id="M96">
<mml:mrow>
<mml:mi>R</mml:mi>
<mml:mi>T</mml:mi>
<mml:mo>~</mml:mo>
<mml:mi>L</mml:mi>
<mml:mo>&#x00B7;</mml:mo>
<mml:mi>E</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>L</mml:mi>
<mml:mo>&#x00B7;</mml:mo>
<mml:mi>&#x0394;</mml:mi>
<mml:mi>S</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>L</mml:mi>
<mml:mo>&#x00B7;</mml:mo>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mi>H</mml:mi>
</mml:msub>
<mml:mo>&#x00B7;</mml:mo>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>E</mml:mi>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:mi mathvariant="normal">(</mml:mi>
<mml:mi>E</mml:mi>
<mml:mi mathvariant="normal">|</mml:mi>
<mml:mi>p</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>. The results of the statistical analysis are reported in <xref ref-type="table" rid="tab2">Table 2</xref>. The regression coefficients, with their respective group significance, are shown in <xref ref-type="fig" rid="fig2">Figures 2E</xref>,<xref ref-type="fig" rid="fig2">F</xref>.</p>
<table-wrap position="float" id="tab2">
<label>Table 2</label>
<caption>
<p>Linear mixed effects model for the percentage of optimal choices selected <inline-formula>
<mml:math id="M97">
<mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi>o</mml:mi>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula> and for the reaction time <inline-formula>
<mml:math id="M98">
<mml:mrow>
<mml:mi>R</mml:mi>
<mml:mi>T</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>.</p>
</caption>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th/>
<th align="center" valign="top">
<inline-formula>
<mml:math id="M99">
<mml:mrow>
<mml:msub>
<mml:mi>P</mml:mi>
<mml:mrow>
<mml:mi>o</mml:mi>
<mml:mi>c</mml:mi>
</mml:mrow>
</mml:msub>
<mml:mo>~</mml:mo>
<mml:mi>L</mml:mi>
<mml:mo>&#x00B7;</mml:mo>
<mml:mi>E</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>L</mml:mi>
<mml:mo>&#x00B7;</mml:mo>
<mml:mi>&#x0394;</mml:mi>
<mml:mi>S</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>L</mml:mi>
<mml:mo>&#x00B7;</mml:mo>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mi>H</mml:mi>
</mml:msub>
<mml:mo>&#x00B7;</mml:mo>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>E</mml:mi>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:mi mathvariant="normal">(</mml:mi>
<mml:mi>E</mml:mi>
<mml:mi mathvariant="normal">|</mml:mi>
<mml:mi>p</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
</th>
<th align="center" valign="top">
<inline-formula>
<mml:math id="M100">
<mml:mrow>
<mml:mi>R</mml:mi>
<mml:mi>T</mml:mi>
<mml:mo>~</mml:mo>
<mml:mi>L</mml:mi>
<mml:mo>&#x00B7;</mml:mo>
<mml:mi>E</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>L</mml:mi>
<mml:mo>&#x00B7;</mml:mo>
<mml:mi>&#x0394;</mml:mi>
<mml:mi>S</mml:mi>
<mml:mo>+</mml:mo>
<mml:mi>L</mml:mi>
<mml:mo>&#x00B7;</mml:mo>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mi>H</mml:mi>
</mml:msub>
<mml:mo>&#x00B7;</mml:mo>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>E</mml:mi>
</mml:msub>
<mml:mo>+</mml:mo>
<mml:mi mathvariant="normal">(</mml:mi>
<mml:mi>E</mml:mi>
<mml:mi mathvariant="normal">|</mml:mi>
<mml:mi>p</mml:mi>
<mml:mo stretchy="false">)</mml:mo>
</mml:mrow>
</mml:math>
</inline-formula>
</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="top">F-stat.</td>
<td align="center" valign="top">175</td>
<td align="center" valign="top">205.9</td>
</tr>
<tr>
<td align="left" valign="top"><italic>p</italic>-value</td>
<td align="center" valign="top">0</td>
<td align="center" valign="top">0</td>
</tr>
</tbody>
</table>
<table frame="hsides" rules="groups">
<thead>
<tr>
<th align="left" valign="top">
<bold>Fixed effects</bold>
</th>
<th align="center" valign="top">
<bold>Estimate</bold>
</th>
<th align="center" valign="top">
<bold>SE</bold>
</th>
<th align="center" valign="top">
<bold><italic>t</italic> Stat</bold>
</th>
<th align="center" valign="top">
<bold><italic>p</italic> Val</bold>
</th>
<th align="center" valign="top">
<bold>Lower</bold>
</th>
<th align="center" valign="top">
<bold>Upper</bold>
</th>
<th align="center" valign="top">
<bold>Estimate</bold>
</th>
<th align="center" valign="top">
<bold>SE</bold>
</th>
<th align="center" valign="top">
<bold><italic>t</italic> Stat</bold>
</th>
<th align="center" valign="top">
<bold><italic>p</italic> Val</bold>
</th>
<th align="center" valign="top">
<bold>Lower</bold>
</th>
<th align="center" valign="top">
<bold>Upper</bold>
</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left" valign="top">Intercept</td>
<td align="center" valign="top">6.35</td>
<td align="center" valign="top">0.40</td>
<td align="center" valign="top">15.6</td>
<td align="center" valign="top">10<sup>&#x2212;54</sup></td>
<td align="center" valign="top">5.55</td>
<td align="center" valign="top">7.15</td>
<td align="center" valign="top">0.75</td>
<td align="center" valign="top">0.15</td>
<td align="center" valign="top">4.95</td>
<td align="center" valign="top">10<sup>&#x2212;07</sup></td>
<td align="center" valign="top">0.456</td>
<td align="center" valign="top">1.05</td>
</tr>
<tr>
<td align="left" valign="top">
<inline-formula>
<mml:math id="M101">
<mml:mrow>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>E</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="center" valign="top">4.38</td>
<td align="center" valign="top">0.26</td>
<td align="center" valign="top">&#x2212;16.9</td>
<td align="center" valign="top">10<sup>&#x2212;64</sup></td>
<td align="center" valign="top">&#x2212;3.88</td>
<td align="center" valign="top">4.89</td>
<td align="center" valign="top">&#x2212;0.58</td>
<td align="center" valign="top">0.08</td>
<td align="center" valign="top">&#x2212;7.04</td>
<td align="center" valign="top">10<sup>&#x2212;12</sup></td>
<td align="center" valign="top">&#x2212;0.75</td>
<td align="center" valign="top">&#x2212;0.42</td>
</tr>
<tr>
<td align="left" valign="top">
<inline-formula>
<mml:math id="M102">
<mml:mrow>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mi>H</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="center" valign="top">&#x2212;1.55</td>
<td align="center" valign="top">0.18</td>
<td align="center" valign="top">&#x2212;8.35</td>
<td align="center" valign="top">10<sup>&#x2212;17</sup></td>
<td align="center" valign="top">&#x2212;1.92</td>
<td align="center" valign="top">&#x2212;1.18</td>
<td align="center" valign="top">&#x2212;0.48</td>
<td align="center" valign="top">0.06</td>
<td align="center" valign="top">&#x2212;8.36</td>
<td align="center" valign="top">10<sup>&#x2212;17</sup></td>
<td align="center" valign="top">&#x2212;0.60</td>
<td align="center" valign="top">&#x2212;0.37</td>
</tr>
<tr>
<td align="left" valign="top">
<italic>E</italic>
</td>
<td align="center" valign="top">&#x2212;0.001</td>
<td align="center" valign="top">0.003</td>
<td align="center" valign="top">&#x2212;0.40</td>
<td align="center" valign="top">0.69</td>
<td align="center" valign="top">&#x2212;0.01</td>
<td align="center" valign="top">0.01</td>
<td align="center" valign="top">&#x2212;0.05</td>
<td align="center" valign="top">0.04</td>
<td align="center" valign="top">&#x2212;1.21</td>
<td align="center" valign="top">0.23</td>
<td align="center" valign="top">&#x2212;0.13</td>
<td align="center" valign="top">0.03</td>
</tr>
<tr>
<td align="left" valign="top">
<inline-formula>
<mml:math id="M103">
<mml:mrow>
<mml:mi>&#x0394;</mml:mi>
<mml:mi>S</mml:mi>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="center" valign="top">&#x2212;1.10</td>
<td align="center" valign="top">0.67</td>
<td align="center" valign="top">&#x2212;1.65</td>
<td align="center" valign="top">0.10</td>
<td align="center" valign="top">&#x2212;2.40</td>
<td align="center" valign="top">0.21</td>
<td align="center" valign="top">&#x2212;0.24</td>
<td align="center" valign="top">0.02</td>
<td align="center" valign="top">&#x2212;15.78</td>
<td align="center" valign="top">10<sup>&#x2212;55</sup></td>
<td align="center" valign="top">&#x2212;0.27</td>
<td align="center" valign="top">&#x2212;0.21</td>
</tr>
<tr>
<td align="left" valign="top">
<italic>L<sub>1</sub></italic>
</td>
<td align="center" valign="top">&#x2212;2.31</td>
<td align="center" valign="top">0.47</td>
<td align="center" valign="top">&#x2212;4.90</td>
<td align="center" valign="top">10<sup>&#x2212;7</sup></td>
<td align="center" valign="top">&#x2212;3.23</td>
<td align="center" valign="top">&#x2212;1.39</td>
<td align="center" valign="top">&#x2212;1.45</td>
<td align="center" valign="top">0.17</td>
<td align="center" valign="top">&#x2212;8.42</td>
<td align="center" valign="top">10<sup>&#x2212;17</sup></td>
<td align="center" valign="top">&#x2212;1.79</td>
<td align="center" valign="top">&#x2212;1.11</td>
</tr>
<tr>
<td align="left" valign="top">
<inline-formula>
<mml:math id="M104">
<mml:mrow>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>E</mml:mi>
</mml:msub>
<mml:mo>:</mml:mo>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mi>H</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="center" valign="top">&#x2212;1.21</td>
<td align="center" valign="top">0.14</td>
<td align="center" valign="top">8.52</td>
<td align="center" valign="top">10<sup>&#x2212;17</sup></td>
<td align="center" valign="top">&#x2212;1.49</td>
<td align="center" valign="top">&#x2212;0.93</td>
<td align="center" valign="top">0.36</td>
<td align="center" valign="top">0.04</td>
<td align="center" valign="top">8.21</td>
<td align="center" valign="top">10<sup>&#x2212;16</sup></td>
<td align="center" valign="top">0.28</td>
<td align="center" valign="top">0.45</td>
</tr>
<tr>
<td align="left" valign="top">
<inline-formula>
<mml:math id="M105">
<mml:mrow>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>E</mml:mi>
</mml:msub>
<mml:mo>:</mml:mo>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="center" valign="top">&#x2212;2.05</td>
<td align="center" valign="top">0.29</td>
<td align="center" valign="top">6.97</td>
<td align="center" valign="top">10<sup>&#x2212;12</sup></td>
<td align="center" valign="top">&#x2212;2.63</td>
<td align="center" valign="top">&#x2212;1.47</td>
<td align="center" valign="top">1.11</td>
<td align="center" valign="top">0.09</td>
<td align="center" valign="top">11.89</td>
<td align="center" valign="top">10<sup>&#x2212;32</sup></td>
<td align="center" valign="top">0.93</td>
<td align="center" valign="top">1.30</td>
</tr>
<tr>
<td align="left" valign="top">
<inline-formula>
<mml:math id="M106">
<mml:mrow>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mi>H</mml:mi>
</mml:msub>
<mml:mo>:</mml:mo>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="center" valign="top">&#x2212;0.08</td>
<td align="center" valign="top">0.21</td>
<td align="center" valign="top">&#x2212;0.37</td>
<td align="center" valign="top">0.71</td>
<td align="center" valign="top">&#x2212;0.51</td>
<td align="center" valign="top">0.35</td>
<td align="center" valign="top">0.62</td>
<td align="center" valign="top">0.07</td>
<td align="center" valign="top">9.48</td>
<td align="center" valign="top">10<sup>&#x2212;21</sup></td>
<td align="center" valign="top">0.49</td>
<td align="center" valign="top">0.75</td>
</tr>
<tr>
<td align="left" valign="top">
<inline-formula>
<mml:math id="M107">
<mml:mrow>
<mml:mi>E</mml:mi>
<mml:mo>:</mml:mo>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="center" valign="top">0.02</td>
<td align="center" valign="top">0.003</td>
<td align="center" valign="top">4.88</td>
<td align="center" valign="top">10<sup>&#x2212;6</sup></td>
<td align="center" valign="top">0.01</td>
<td align="center" valign="top">0.02</td>
<td align="center" valign="top">&#x2212;0.02</td>
<td align="center" valign="top">0.04</td>
<td align="center" valign="top">&#x2212;0.49</td>
<td align="center" valign="top">0.62</td>
<td align="center" valign="top">&#x2212;0.11</td>
<td align="center" valign="top">0.07</td>
</tr>
<tr>
<td align="left" valign="top">
<inline-formula>
<mml:math id="M108">
<mml:mrow>
<mml:mi>&#x0394;</mml:mi>
<mml:mi>S</mml:mi>
<mml:mo>:</mml:mo>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="center" valign="top">8.61</td>
<td align="center" valign="top">0.78</td>
<td align="center" valign="top">11.08</td>
<td align="center" valign="top">10<sup>&#x2212;28</sup></td>
<td align="center" valign="top">7.08</td>
<td align="center" valign="top">10.13</td>
<td align="center" valign="top">&#x2212;0.06</td>
<td align="center" valign="top">0.02</td>
<td align="center" valign="top">&#x2212;3.42</td>
<td align="center" valign="top">10<sup>&#x2212;3</sup></td>
<td align="center" valign="top">&#x2212;0.09</td>
<td align="center" valign="top">&#x2212;0.02</td>
</tr>
<tr>
<td align="left" valign="top">
<inline-formula>
<mml:math id="M109">
<mml:mrow>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>E</mml:mi>
</mml:msub>
<mml:mo>:</mml:mo>
<mml:msub>
<mml:mi>n</mml:mi>
<mml:mi>H</mml:mi>
</mml:msub>
<mml:mo>:</mml:mo>
<mml:msub>
<mml:mi>L</mml:mi>
<mml:mn>1</mml:mn>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>
</td>
<td align="center" valign="top">&#x2212;0.22</td>
<td align="center" valign="top">0.16</td>
<td align="center" valign="top">&#x2212;1.40</td>
<td align="center" valign="top">0.16</td>
<td align="center" valign="top">&#x2212;0.54</td>
<td align="center" valign="top">0.09</td>
<td align="center" valign="top">&#x2212;0.51</td>
<td align="center" valign="top">0.05</td>
<td align="center" valign="top">&#x2212;10.31</td>
<td align="center" valign="top">10<sup>&#x2212;25</sup></td>
<td align="center" valign="top">&#x2212;0.61</td>
<td align="center" valign="top">&#x2212;0.42</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p>The independent variables for the fixed effects are horizon n<sub>H</sub>, trial within episode <inline-formula>
<mml:math id="M110">
<mml:mrow>
<mml:msub>
<mml:mi>T</mml:mi>
<mml:mi>E</mml:mi>
</mml:msub>
</mml:mrow>
</mml:math>
</inline-formula>, and the passage of time expressed as episodes E, and &#x0394;S. We set the random effects for the intercept and the episodes grouped by participant <italic>p</italic>.</p>
</table-wrap-foot>
</table-wrap>
</sec>
</sec>
<sec sec-type="data-availability" id="sec27">
<title>Data availability statement</title>
<p>The datasets generated during and analyzed during the current study are available in the eBrains repository, <ext-link xlink:href="https://search.kg.ebrains.eu/instances/0d145ebe-3ecd-4b3c-9400-913a8cd21a6a" ext-link-type="uri">https://search.kg.ebrains.eu/instances/0d145ebe-3ecd-4b3c-9400-913a8cd21a6a</ext-link>. The codes generated during the current study are available in the eBrains repository <ext-link xlink:href="https://search.kg.ebrains.eu/instances/ffda985e-9023-4d06-aa79-0ec7109ff55c" ext-link-type="uri">https://search.kg.ebrains.eu/instances/ffda985e-9023-4d06-aa79-0ec7109ff55c</ext-link> linked to the GitHub repository <ext-link xlink:href="https://github.com/gloriacec/Model_ConsequenceBasedDecisionMaking" ext-link-type="uri">https://github.com/gloriacec/Model_ConsequenceBasedDecisionMaking</ext-link>.</p>
</sec>
<sec sec-type="ethics-statement" id="sec28">
<title>Ethics statement</title>
<p>The studies involving humans were approved by Clinical Research Ethics Committee (CEIm Ref. #2021/9743/I). The studies were conducted in accordance with the local legislation and institutional requirements. The ethics committee/institutional review board waived the requirement of written informed consent for participation from the participants or the participants&#x2019; legal guardians/next of kin because all participants were neurologically healthy, had normal or corrected to normal vision, were naive as to the purpose of the study, and gave informed consent before participating.</p>
</sec>
<sec sec-type="author-contributions" id="sec29">
<title>Author contributions</title>
<p>GC: Conceptualization, Data curation, Formal analysis, Investigation, Methodology, Software, Validation, Visualization, Writing &#x2013; original draft, Writing &#x2013; review &#x0026; editing. MD: Data curation, Investigation, Methodology, Writing &#x2013; review &#x0026; editing. EB: Investigation, Writing &#x2013; review &#x0026; editing. MA: Investigation, Writing &#x2013; review &#x0026; editing. SR: Writing &#x2013; review &#x0026; editing. PP: Writing &#x2013; review &#x0026; editing. SF: Funding acquisition, Supervision, Writing &#x2013; review &#x0026; editing. AD: Funding acquisition, Supervision, Writing &#x2013; review &#x0026; editing. RM-B: Funding acquisition, Supervision, Writing &#x2013; review &#x0026; editing. IC: Funding acquisition, Investigation, Methodology, Supervision, Writing &#x2013; original draft, Writing &#x2013; review &#x0026; editing.</p>
</sec>
</body>
<back>
<sec sec-type="funding-information" id="sec30">
<title>Funding</title>
<p>The author(s) declare that financial support was received for the research, authorship, and/or publication of this article. This project has received funding from the European Union&#x2019;s Horizon 2020 Framework Programme for Research and Innovation under the Specific Grant Agreement N. 945539 COREDEM (Human Brain Project SGA3).</p>
</sec>
<sec sec-type="COI-statement" id="sec31">
<title>Conflict of interest</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p>
<p>The author(s) declared that they were an editorial board member of Frontiers, at the time of submission. This had no impact on the peer review process and the final decision.</p>
</sec>
<sec sec-type="disclaimer" id="sec32">
<title>Publisher&#x2019;s note</title>
<p>All claims expressed in this article are solely those of the authors and do not necessarily represent those of their affiliated organizations, or those of the publisher, the editors and the reviewers. Any product that may be evaluated in this article, or claim that may be made by its manufacturer, is not guaranteed or endorsed by the publisher.</p>
</sec>
<sec sec-type="supplementary-material" id="sec33">
<title>Supplementary material</title>
<p>The Supplementary material for this article can be found online at: <ext-link xlink:href="https://www.frontiersin.org/articles/10.3389/fnbeh.2024.1399394/full#supplementary-material" ext-link-type="uri">https://www.frontiersin.org/articles/10.3389/fnbeh.2024.1399394/full#supplementary-material</ext-link></p>
<supplementary-material xlink:href="Data_Sheet_1.pdf" id="SM1" mimetype="application/pdf" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<ref-list>
<title>References</title>
<ref id="ref1">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Alexander</surname> <given-names>W. H.</given-names></name> <name><surname>Brown</surname> <given-names>J. W.</given-names></name></person-group> (<year>2010</year>). <article-title>Hyperbolically discounted temporal difference learning</article-title>. <source>Neural Comput.</source> <volume>22</volume>, <fpage>1511</fpage>&#x2013;<lpage>1527</lpage>. doi: <pub-id pub-id-type="doi">10.1162/neco.2010.08-09-1080</pub-id>, PMID: <pub-id pub-id-type="pmid">20100071</pub-id></citation>
</ref>
<ref id="ref2">
<citation citation-type="journal"><person-group person-group-type="author">
<name><surname>Amari</surname> <given-names>S. I.</given-names></name>
</person-group> (<year>1998</year>). <article-title>Natural gradient works efficiently in learning</article-title>. <source>Neural Comput.</source> <volume>10</volume>, <fpage>251</fpage>&#x2013;<lpage>276</lpage>. doi: <pub-id pub-id-type="doi">10.1162/089976698300017746</pub-id>, PMID: <pub-id pub-id-type="pmid">38998680</pub-id></citation>
</ref>
<ref id="ref3">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Apps</surname> <given-names>M. A. J.</given-names></name> <name><surname>Grima</surname> <given-names>L. L.</given-names></name> <name><surname>Manohar</surname> <given-names>S.</given-names></name> <name><surname>Husain</surname> <given-names>M.</given-names></name></person-group> (<year>2015</year>). <article-title>The role of cognitive effort in subjective reward devaluation and risky decision-making</article-title>. <source>Scientific Reports 2015 5: 1</source> <volume>5</volume>, <fpage>1</fpage>&#x2013;<lpage>11</lpage>. doi: <pub-id pub-id-type="doi">10.1038/srep16880</pub-id></citation>
</ref>
<ref id="ref4">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Balasubramani</surname> <given-names>P. P.</given-names></name> <name><surname>Hayden</surname> <given-names>B. Y.</given-names></name></person-group> (<year>2018</year>). <article-title>Overlapping neural processes for stopping and economic choice in orbitofrontal cortex</article-title>. <source>bio Rxiv</source>:<fpage>304709</fpage>. doi: <pub-id pub-id-type="doi">10.1101/304709</pub-id></citation>
</ref>
<ref id="ref5">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Barbosa</surname> <given-names>J.</given-names></name> <name><surname>Stein</surname> <given-names>H.</given-names></name> <name><surname>Martinez</surname> <given-names>R. L.</given-names></name> <name><surname>Galan-Gadea</surname> <given-names>A.</given-names></name> <name><surname>Li</surname> <given-names>S.</given-names></name> <name><surname>Dalmau</surname> <given-names>J.</given-names></name> <etal/></person-group>. (<year>2020</year>). <article-title>Interplay between persistent activity and activity-silent dynamics in the prefrontal cortex underlies serial biases in working memory</article-title>. <source>Nat. Neurosci.</source> <volume>23</volume>, <fpage>1016</fpage>&#x2013;<lpage>1024</lpage>. doi: <pub-id pub-id-type="doi">10.1038/s41593-020-0644-4</pub-id>, PMID: <pub-id pub-id-type="pmid">32572236</pub-id></citation>
</ref>
<ref id="ref6">
<citation citation-type="journal"><person-group person-group-type="author">
<name><surname>Birnbaum</surname> <given-names>M. H.</given-names></name>
</person-group> (<year>2008</year>). <article-title>New paradoxes of risky decision making</article-title>. <source>Psychol. Rev.</source> <volume>115</volume>, <fpage>463</fpage>&#x2013;<lpage>501</lpage>. doi: <pub-id pub-id-type="doi">10.1037/0033-295X.115.2.463</pub-id>, PMID: <pub-id pub-id-type="pmid">18426300</pub-id></citation>
</ref>
<ref id="ref7">
<citation citation-type="journal"><person-group person-group-type="author">
<name><surname>Blake</surname> <given-names>R.</given-names></name>
</person-group> (<year>1989</year>). <article-title>A neural theory of binocular rivalry</article-title>. <source>Psychol. Rev.</source> <volume>96</volume>, <fpage>145</fpage>&#x2013;<lpage>167</lpage>. doi: <pub-id pub-id-type="doi">10.1037/0033-295X.96.1.145</pub-id>, PMID: <pub-id pub-id-type="pmid">2648445</pub-id></citation>
</ref>
<ref id="ref8">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Boelts</surname> <given-names>J.</given-names></name> <name><surname>Lueckmann</surname> <given-names>J.-M.</given-names></name> <name><surname>Gao</surname> <given-names>R.</given-names></name> <name><surname>Macke</surname> <given-names>J. H.</given-names></name></person-group> (<year>2022</year>). <article-title>Flexible and efficient simulation-based inference for models of decision-making</article-title>. <source>eLife</source> <volume>11</volume>:<fpage>e77220</fpage>. doi: <pub-id pub-id-type="doi">10.7554/eLife.77220</pub-id>, PMID: <pub-id pub-id-type="pmid">35894305</pub-id></citation>
</ref>
<ref id="ref9">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Britten</surname> <given-names>K. H.</given-names></name> <name><surname>Shadlen</surname> <given-names>M. N.</given-names></name> <name><surname>Newsome</surname> <given-names>W. T.</given-names></name> <name><surname>Movshon</surname> <given-names>J. A.</given-names></name></person-group> (<year>1993</year>). <article-title>Responses of neurons in macaque MT to stochastic motion signals</article-title>. <source>Vis. Neurosci.</source> <volume>10</volume>, <fpage>1157</fpage>&#x2013;<lpage>1169</lpage>. doi: <pub-id pub-id-type="doi">10.1017/S0952523800010269</pub-id>, PMID: <pub-id pub-id-type="pmid">8257671</pub-id></citation>
</ref>
<ref id="ref10">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Brunel</surname> <given-names>N.</given-names></name> <name><surname>Wang</surname> <given-names>X. J.</given-names></name></person-group> (<year>2001</year>). <article-title>Effects of Neuromodulation in a cortical network model of object working memory dominated by recurrent inhibition</article-title>. <source>J. Comput. Neurosci.</source> <volume>11</volume>, <fpage>63</fpage>&#x2013;<lpage>85</lpage>. doi: <pub-id pub-id-type="doi">10.1023/A:1011204814320</pub-id>, PMID: <pub-id pub-id-type="pmid">11524578</pub-id></citation>
</ref>
<ref id="ref11">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Cai</surname> <given-names>X.</given-names></name> <name><surname>Padoa-Schioppa</surname> <given-names>C.</given-names></name></person-group> (<year>2019</year>). <article-title>Neuronal evidence for good-based economic decisions under variable action costs</article-title>. <source>Nat. Commun.</source> <volume>10</volume>, <fpage>1</fpage>&#x2013;<lpage>13</lpage>. doi: <pub-id pub-id-type="doi">10.1038/s41467-018-08209-3</pub-id></citation>
</ref>
<ref id="ref12">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Carroll</surname> <given-names>T. J.</given-names></name> <name><surname>McNamee</surname> <given-names>D.</given-names></name> <name><surname>Ingram</surname> <given-names>J. N.</given-names></name> <name><surname>Wolpert</surname> <given-names>D. M.</given-names></name></person-group> (<year>2019</year>). <article-title>Rapid Visuomotor responses reflect value-based decisions</article-title>. <source>J. Neurosci.</source> <volume>39</volume>, <fpage>3906</fpage>&#x2013;<lpage>3920</lpage>. doi: <pub-id pub-id-type="doi">10.1523/JNEUROSCI.1934-18.2019</pub-id>, PMID: <pub-id pub-id-type="pmid">30850511</pub-id></citation>
</ref>
<ref id="ref13">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Cavanagh</surname> <given-names>S. E.</given-names></name> <name><surname>Towers</surname> <given-names>J. P.</given-names></name> <name><surname>Wallis</surname> <given-names>J. D.</given-names></name> <name><surname>Hunt</surname> <given-names>L. T.</given-names></name> <name><surname>Kennerley</surname> <given-names>S. W.</given-names></name></person-group> (<year>2018</year>). <article-title>Reconciling persistent and dynamic hypotheses of working memory coding in prefrontal cortex</article-title>. <source>Nat. Commun.</source> <volume>9</volume>, <fpage>1</fpage>&#x2013;<lpage>16</lpage>. doi: <pub-id pub-id-type="doi">10.1038/s41467-018-05873-3</pub-id></citation>
</ref>
<ref id="ref14">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Cisek</surname> <given-names>P.</given-names></name> <name><surname>Kalaska</surname> <given-names>J. F.</given-names></name></person-group> (<year>2005</year>). <article-title>Neural correlates of reaching decisions in dorsal premotor cortex: specification of multiple direction choices and final selection of action</article-title>. <source>Neuron</source> <volume>45</volume>, <fpage>801</fpage>&#x2013;<lpage>814</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.neuron.2005.01.027</pub-id>, PMID: <pub-id pub-id-type="pmid">15748854</pub-id></citation>
</ref>
<ref id="ref15">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Cisek</surname> <given-names>P.</given-names></name> <name><surname>Puskas</surname> <given-names>G. A.</given-names></name> <name><surname>El-Murr</surname> <given-names>S.</given-names></name></person-group> (<year>2009</year>). <article-title>Decisions in changing conditions: the urgency-gating model</article-title>. <source>J. Neurosci.</source> <volume>29</volume>, <fpage>11560</fpage>&#x2013;<lpage>11571</lpage>. doi: <pub-id pub-id-type="doi">10.1523/JNEUROSCI.1844-09.2009</pub-id>, PMID: <pub-id pub-id-type="pmid">19759303</pub-id></citation>
</ref>
<ref id="ref16">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Cos</surname> <given-names>I.</given-names></name> <name><surname>Khamassi</surname> <given-names>M.</given-names></name> <name><surname>Girard</surname> <given-names>B.</given-names></name></person-group> (<year>2013</year>). <article-title>Modelling the learning of biomechanics and visual planning for decision-making of motor actions</article-title>. <source>J.Physiol. Paris</source> <volume>107</volume>, <fpage>399</fpage>&#x2013;<lpage>408</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.jphysparis.2013.07.004</pub-id>, PMID: <pub-id pub-id-type="pmid">23973913</pub-id></citation>
</ref>
<ref id="ref17">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Danwitz</surname> <given-names>L.</given-names></name> <name><surname>Mathar</surname> <given-names>D.</given-names></name> <name><surname>Smith</surname> <given-names>E.</given-names></name> <name><surname>Tuzsus</surname> <given-names>D.</given-names></name> <name><surname>Peters</surname> <given-names>J.</given-names></name></person-group> (<year>2022</year>). <article-title>Parameter and model recovery of reinforcement learning models for restless bandit problems</article-title>. <source>Comput. Brain Behav.</source> <volume>5</volume>, <fpage>547</fpage>&#x2013;<lpage>563</lpage>. doi: <pub-id pub-id-type="doi">10.1007/s42113-022-00139-0</pub-id></citation>
</ref>
<ref id="ref18">
<citation citation-type="journal"><person-group person-group-type="author">
<name><surname>Dayan</surname> <given-names>P.</given-names></name>
</person-group> (<year>1992</year>). <article-title>The convergence of TD(&#x03BB;) for general &#x03BB;</article-title>. <source>Mach. Learn.</source> <volume>8</volume>, <fpage>341</fpage>&#x2013;<lpage>362</lpage>. doi: <pub-id pub-id-type="doi">10.1023/A:1022632907294/METRICS</pub-id></citation>
</ref>
<ref id="ref19">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Deco</surname> <given-names>G.</given-names></name> <name><surname>Rolls</surname> <given-names>E. T.</given-names></name></person-group> (<year>2005</year>). <article-title>Attention, short-term memory, and action selection: a unifying theory</article-title>. <source>Prog. Neurobiol.</source> <volume>76</volume>, <fpage>236</fpage>&#x2013;<lpage>256</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.pneurobio.2005.08.004</pub-id>, PMID: <pub-id pub-id-type="pmid">16257103</pub-id></citation>
</ref>
<ref id="ref20">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Donner</surname> <given-names>T. H.</given-names></name> <name><surname>Siegel</surname> <given-names>M.</given-names></name> <name><surname>Fries</surname> <given-names>P.</given-names></name> <name><surname>Engel</surname> <given-names>A. K.</given-names></name></person-group> (<year>2009</year>). <article-title>Buildup of choice-predictive activity in human motor cortex during perceptual decision making</article-title>. <source>Curr. Biol.</source> <volume>19</volume>, <fpage>1581</fpage>&#x2013;<lpage>1585</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.cub.2009.07.066</pub-id>, PMID: <pub-id pub-id-type="pmid">19747828</pub-id></citation>
</ref>
<ref id="ref21">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Drugowitsch</surname> <given-names>J.</given-names></name> <name><surname>Moreno-Bote</surname> <given-names>R. N.</given-names></name> <name><surname>Churchland</surname> <given-names>A. K.</given-names></name> <name><surname>Shadlen</surname> <given-names>M. N.</given-names></name> <name><surname>Pouget</surname> <given-names>A.</given-names></name></person-group> (<year>2012</year>). <article-title>The cost of accumulating evidence in perceptual decision making</article-title>. <source>J. Neurosci.</source> <volume>32</volume>, <fpage>3612</fpage>&#x2013;<lpage>3628</lpage>. doi: <pub-id pub-id-type="doi">10.1523/JNEUROSCI.4010-11.2012</pub-id>, PMID: <pub-id pub-id-type="pmid">22423085</pub-id></citation>
</ref>
<ref id="ref22">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Drugowitsch</surname> <given-names>J.</given-names></name> <name><surname>Wyart</surname> <given-names>V.</given-names></name> <name><surname>Devauchelle</surname> <given-names>A. D.</given-names></name> <name><surname>Koechlin</surname> <given-names>E.</given-names></name></person-group> (<year>2016</year>). <article-title>Computational precision of mental inference as critical source of human choice suboptimality</article-title>. <source>Neuron</source> <volume>92</volume>, <fpage>1398</fpage>&#x2013;<lpage>1411</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.neuron.2016.11.005</pub-id>, PMID: <pub-id pub-id-type="pmid">27916454</pub-id></citation>
</ref>
<ref id="ref23">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Eichberger</surname> <given-names>J.</given-names></name> <name><surname>Pasichnichenko</surname> <given-names>I.</given-names></name></person-group> (<year>2021</year>). <article-title>Decision-making with partial information</article-title>. <source>J. Econ. Theory</source> <volume>198</volume>:<fpage>105369</fpage>. doi: <pub-id pub-id-type="doi">10.1016/j.jet.2021.105369</pub-id>, PMID: <pub-id pub-id-type="pmid">39016111</pub-id></citation>
</ref>
<ref id="ref24">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Evans</surname> <given-names>N. J.</given-names></name> <name><surname>Trueblood</surname> <given-names>J. S.</given-names></name> <name><surname>Holmes</surname> <given-names>W. R.</given-names></name></person-group> (<year>2020</year>). <article-title>A parameter recovery assessment of time-variant models of decision-making</article-title>. <source>Behav. Res. Methods</source> <volume>52</volume>, <fpage>193</fpage>&#x2013;<lpage>206</lpage>. doi: <pub-id pub-id-type="doi">10.3758/s13428-019-01218-0</pub-id>, PMID: <pub-id pub-id-type="pmid">30924107</pub-id></citation>
</ref>
<ref id="ref25">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Fontanesi</surname> <given-names>L.</given-names></name> <name><surname>Gluth</surname> <given-names>S.</given-names></name> <name><surname>Spektor</surname> <given-names>M. S.</given-names></name> <name><surname>Rieskamp</surname> <given-names>J.</given-names></name></person-group> (<year>2019</year>). <article-title>A reinforcement learning diffusion decision model for value-based decisions</article-title>. <source>Psychon. Bull. Rev.</source> <volume>26</volume>, <fpage>1099</fpage>&#x2013;<lpage>1121</lpage>. doi: <pub-id pub-id-type="doi">10.3758/s13423-018-1554-2</pub-id>, PMID: <pub-id pub-id-type="pmid">30924057</pub-id></citation>
</ref>
<ref id="ref26">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Ga&#x0142;ecki</surname> <given-names>A.</given-names></name> <name><surname>Burzykowski</surname> <given-names>T.</given-names></name></person-group> (<year>2013</year>). <source><italic>Linear mixed-effects models using R: A step-by-step approach</italic>. In springer texts in statistics</source>. <publisher-loc>New York</publisher-loc>: <publisher-name>Springer</publisher-name>.</citation>
</ref>
<ref id="ref27">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gluth</surname> <given-names>S.</given-names></name> <name><surname>Rieskamp</surname> <given-names>J.</given-names></name> <name><surname>B&#x00FC;chel</surname> <given-names>C.</given-names></name></person-group> (<year>2014</year>). <article-title>Neural evidence for adaptive strategy selection in value-based decision-making</article-title>. <source>Cereb. Cortex</source> <volume>24</volume>, <fpage>2009</fpage>&#x2013;<lpage>2021</lpage>. doi: <pub-id pub-id-type="doi">10.1093/cercor/bht049</pub-id>, PMID: <pub-id pub-id-type="pmid">23476024</pub-id></citation>
</ref>
<ref id="ref28">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gold</surname> <given-names>J. I.</given-names></name> <name><surname>Shadlen</surname> <given-names>M. N.</given-names></name></person-group> (<year>2007</year>). <article-title>The neural basis of decision making</article-title>. <source>Annu. Rev. Neurosci.</source> <volume>30</volume>, <fpage>535</fpage>&#x2013;<lpage>574</lpage>. doi: <pub-id pub-id-type="doi">10.1146/ANNUREV.NEURO.29.051605.113038</pub-id></citation>
</ref>
<ref id="ref29">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Goodwin</surname> <given-names>S. J.</given-names></name> <name><surname>Blackman</surname> <given-names>R. K.</given-names></name> <name><surname>Sakellaridi</surname> <given-names>S.</given-names></name> <name><surname>Chafee</surname> <given-names>M. V.</given-names></name></person-group> (<year>2012</year>). <article-title>Executive control over cognition: stronger and earlier rule-based modulation of spatial category signals in prefrontal cortex relative to parietal cortex</article-title>. <source>J. Neurosci.</source> <volume>32</volume>, <fpage>3499</fpage>&#x2013;<lpage>3515</lpage>. doi: <pub-id pub-id-type="doi">10.1523/JNEUROSCI.3585-11.2012</pub-id>, PMID: <pub-id pub-id-type="pmid">22399773</pub-id></citation>
</ref>
<ref id="ref30">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gureckis</surname> <given-names>T. M.</given-names></name> <name><surname>Love</surname> <given-names>B. C.</given-names></name></person-group> (<year>2009</year>). <article-title>Short-term gains, long-term pains: how cues about state aid learning in dynamic environments</article-title>. <source>Cognition</source> <volume>113</volume>, <fpage>293</fpage>&#x2013;<lpage>313</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.cognition.2009.03.013</pub-id></citation>
</ref>
<ref id="ref31">
<citation citation-type="journal"><person-group person-group-type="author">
<name><surname>Hayden</surname> <given-names>B. Y.</given-names></name>
</person-group> (<year>2016</year>). <article-title>Time discounting and time preference in animals: a critical review</article-title>. <source>Psychon. Bull. Rev.</source> <volume>23</volume>, <fpage>39</fpage>&#x2013;<lpage>53</lpage>. doi: <pub-id pub-id-type="doi">10.3758/s13423-015-0879-3</pub-id>, PMID: <pub-id pub-id-type="pmid">26063653</pub-id></citation>
</ref>
<ref id="ref32">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hayden</surname> <given-names>B. Y.</given-names></name> <name><surname>Platt</surname> <given-names>M. L.</given-names></name></person-group> (<year>2007</year>). <article-title>Temporal discounting predicts risk sensitivity in rhesus macaques</article-title>. <source>Curr. Biol.</source> <volume>17</volume>, <fpage>49</fpage>&#x2013;<lpage>53</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.cub.2006.10.055</pub-id>, PMID: <pub-id pub-id-type="pmid">17208186</pub-id></citation>
</ref>
<ref id="ref33">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hern&#x00E1;ndez</surname> <given-names>A.</given-names></name> <name><surname>N&#x00E1;cher</surname> <given-names>V.</given-names></name> <name><surname>Luna</surname> <given-names>R.</given-names></name> <name><surname>Zainos</surname> <given-names>A.</given-names></name> <name><surname>Lemus</surname> <given-names>L.</given-names></name> <name><surname>Alvarez</surname> <given-names>M.</given-names></name> <etal/></person-group>. (<year>2010</year>). <article-title>Decoding a perceptual decision process across cortex</article-title>. <source>Neuron</source> <volume>66</volume>, <fpage>300</fpage>&#x2013;<lpage>314</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.neuron.2010.03.031</pub-id>, PMID: <pub-id pub-id-type="pmid">20435005</pub-id></citation>
</ref>
<ref id="ref34">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hert&#x00E4;g</surname> <given-names>L.</given-names></name> <name><surname>Durstewitz</surname> <given-names>D.</given-names></name> <name><surname>Brunel</surname> <given-names>N.</given-names></name></person-group> (<year>2014</year>). <article-title>Analytical approximations of the firing rate of an adaptive exponential integrate-and-fire neuron in the presence of synaptic noise</article-title>. <source>Front. Comput. Neurosci.</source> <volume>8</volume>:<fpage>116</fpage>. doi: <pub-id pub-id-type="doi">10.3389/fncom.2014.00116</pub-id>, PMID: <pub-id pub-id-type="pmid">25278872</pub-id></citation>
</ref>
<ref id="ref35">
<citation citation-type="other"><person-group person-group-type="author"><name><surname>Houk</surname> <given-names>J. C.</given-names></name> <name><surname>Adams</surname> <given-names>J. L.</given-names></name> <name><surname>Barto</surname> <given-names>A. G.</given-names></name></person-group> (<year>1995</year>). &#x201C;<article-title>A model of how the basal ganglia generate and use neural signals that predict reinforcement</article-title>&#x201D; in <source>Models of information processing in the basal ganglia</source>, Eds. <person-group person-group-type="editor"><name><surname>Houk</surname> <given-names>J. C.</given-names></name> <name><surname>Davis</surname> <given-names>J. L.</given-names></name> <name><surname>Beiser</surname> <given-names>D. G.</given-names></name></person-group>. <publisher-name>The MIT Press</publisher-name>. <fpage>249</fpage>&#x2013;<lpage>270</lpage>.</citation>
</ref>
<ref id="ref36">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Huber-Carol</surname> <given-names>C.</given-names></name> <name><surname>Balakrishnan</surname> <given-names>N.</given-names></name> <name><surname>Nikulin</surname> <given-names>M. S.</given-names></name> <name><surname>Mesbah</surname> <given-names>M.</given-names></name></person-group> (<year>2002</year>). <source><italic>Goodness-of-fit tests and model validity</italic>, 1</source>. <publisher-loc>Boston</publisher-loc>: <publisher-name>Birkh&#x00E4;user</publisher-name>.</citation>
</ref>
<ref id="ref37">
<citation citation-type="other"><person-group person-group-type="author"><name><surname>Huber-Carol</surname> <given-names>C.</given-names></name> <name><surname>Nikulin</surname> <given-names>M.</given-names></name> <name><surname>Nikulin</surname> <given-names>M. S.</given-names></name> <name><surname>Chimitova</surname> <given-names>E. V</given-names></name></person-group>, &#x201C;Chi-squared goodness-of-fit tests for censored data,&#x201D; <italic>chi-squared goodness-of-fit tests for censored data</italic>. <publisher-loc>Wiley</publisher-loc> (<year>2017</year>).</citation>
</ref>
<ref id="ref38">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hwang</surname> <given-names>J.</given-names></name> <name><surname>Kim</surname> <given-names>S.</given-names></name> <name><surname>Lee</surname> <given-names>D.</given-names></name></person-group> (<year>2009</year>). <article-title>Temporal discounting and inter-temporal choice in rhesus monkeys</article-title>. <source>Front. Behav. Neurosci.</source> <volume>3</volume>:<fpage>567</fpage>. doi: <pub-id pub-id-type="doi">10.3389/neuro.08.009.2009</pub-id>, PMID: <pub-id pub-id-type="pmid">19562091</pub-id></citation>
</ref>
<ref id="ref39">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hyafil</surname> <given-names>A.</given-names></name> <name><surname>Moreno-Bote</surname> <given-names>R.</given-names></name></person-group> (<year>2017</year>). <article-title>Breaking down hierarchies of decision-making in primates</article-title>. <source>eLife</source> <volume>6</volume>:<fpage>e16650</fpage>. doi: <pub-id pub-id-type="doi">10.7554/eLife.16650</pub-id>, PMID: <pub-id pub-id-type="pmid">28648171</pub-id></citation>
</ref>
<ref id="ref40">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kaelbling</surname> <given-names>L. P.</given-names></name> <name><surname>Littman</surname> <given-names>M. L.</given-names></name> <name><surname>Cassandra</surname> <given-names>A. R.</given-names></name></person-group> (<year>1998</year>). <article-title>Planning and acting in partially observable stochastic domains</article-title>. <source>Artif. Intell.</source> <volume>101</volume>, <fpage>99</fpage>&#x2013;<lpage>134</lpage>. doi: <pub-id pub-id-type="doi">10.1016/S0004-3702(98)00023-X</pub-id></citation>
</ref>
<ref id="ref41">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kahneman</surname> <given-names>D.</given-names></name> <name><surname>Tversky</surname> <given-names>A.</given-names></name></person-group> (<year>1979</year>). <article-title>Prospect theory: an analysis of decision under risk</article-title>. <source>Econometria</source> <volume>47</volume>, <fpage>263</fpage>&#x2013;<lpage>292</lpage>. doi: <pub-id pub-id-type="doi">10.2307/1914185</pub-id></citation>
</ref>
<ref id="ref42">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kilpatrick</surname> <given-names>Z. P.</given-names></name> <name><surname>Holmes</surname> <given-names>W. R.</given-names></name> <name><surname>Eissa</surname> <given-names>T. L.</given-names></name> <name><surname>Josi&#x0107;</surname> <given-names>K.</given-names></name></person-group> (<year>2019</year>). <article-title>Optimal models of decision-making in dynamic environments</article-title>. <source>Curr. Opin. Neurobiol.</source> <volume>58</volume>, <fpage>54</fpage>&#x2013;<lpage>60</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.conb.2019.06.006</pub-id>, PMID: <pub-id pub-id-type="pmid">31326724</pub-id></citation>
</ref>
<ref id="ref43">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kim</surname> <given-names>S.</given-names></name> <name><surname>Hwang</surname> <given-names>J.</given-names></name> <name><surname>Lee</surname> <given-names>D.</given-names></name></person-group> (<year>2008</year>). <article-title>Prefrontal coding of temporally discounted values during intertemporal choice</article-title>. <source>Neuron</source> <volume>59</volume>, <fpage>161</fpage>&#x2013;<lpage>172</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.neuron.2008.05.010</pub-id>, PMID: <pub-id pub-id-type="pmid">18614037</pub-id></citation>
</ref>
<ref id="ref44">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kirchler</surname> <given-names>M.</given-names></name> <name><surname>Andersson</surname> <given-names>D.</given-names></name> <name><surname>Bonn</surname> <given-names>C.</given-names></name> <name><surname>Johannesson</surname> <given-names>M.</given-names></name> <name><surname>S&#x00F8;rensen</surname> <given-names>E. &#x00D8;</given-names></name> <name><surname>Stefan</surname> <given-names>M.</given-names></name> <etal/></person-group>. (<year>2017</year>). <article-title>The effect of fast and slow decisions on risk taking</article-title>. <source>J. Risk Uncertain.</source> <volume>54</volume>, <fpage>37</fpage>&#x2013;<lpage>59</lpage>. doi: <pub-id pub-id-type="doi">10.1007/s11166-017-9252-4</pub-id>, PMID: <pub-id pub-id-type="pmid">28725117</pub-id></citation>
</ref>
<ref id="ref45">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Klaes</surname> <given-names>C.</given-names></name> <name><surname>Westendorff</surname> <given-names>S.</given-names></name> <name><surname>Chakrabarti</surname> <given-names>S.</given-names></name> <name><surname>Gail</surname> <given-names>A.</given-names></name></person-group> (<year>2011</year>). <article-title>Choosing goals, not rules: deciding among rule-based action plans</article-title>. <source>Neuron</source> <volume>70</volume>, <fpage>536</fpage>&#x2013;<lpage>548</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.neuron.2011.02.053</pub-id>, PMID: <pub-id pub-id-type="pmid">21555078</pub-id></citation>
</ref>
<ref id="ref46">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Krajbich</surname> <given-names>I.</given-names></name> <name><surname>Armel</surname> <given-names>C.</given-names></name> <name><surname>Rangel</surname> <given-names>A.</given-names></name></person-group> (<year>2010</year>). <article-title>Visual fixations and the computation and comparison of value in simple choice</article-title>. <source>Nat. Neurosci.</source> <volume>13</volume>, <fpage>1292</fpage>&#x2013;<lpage>1298</lpage>. doi: <pub-id pub-id-type="doi">10.1038/nn.2635</pub-id>, PMID: <pub-id pub-id-type="pmid">20835253</pub-id></citation>
</ref>
<ref id="ref47">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Krajbich</surname> <given-names>I.</given-names></name> <name><surname>Rangel</surname> <given-names>A.</given-names></name></person-group> (<year>2011</year>). <article-title>Multialternative drift-diffusion model predicts the relationship between visual fixations and choice in value-based decisions</article-title>. <source>Proc. Natl. Acad. Sci. USA</source> <volume>108</volume>, <fpage>13852</fpage>&#x2013;<lpage>13857</lpage>. doi: <pub-id pub-id-type="doi">10.1073/pnas.1101328108</pub-id>, PMID: <pub-id pub-id-type="pmid">21808009</pub-id></citation>
</ref>
<ref id="ref48">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kurniawan</surname> <given-names>I. T.</given-names></name> <name><surname>Guitart-Masip</surname> <given-names>M.</given-names></name> <name><surname>Dayan</surname> <given-names>P.</given-names></name> <name><surname>Dolan</surname> <given-names>R. J.</given-names></name></person-group> (<year>2013</year>). <article-title>Effort and valuation in the brain: the effects of anticipation and execution</article-title>. <source>J. Neurosci.</source> <volume>33</volume>, <fpage>6160</fpage>&#x2013;<lpage>6169</lpage>. doi: <pub-id pub-id-type="doi">10.1523/JNEUROSCI.4777-12.2013</pub-id>, PMID: <pub-id pub-id-type="pmid">23554497</pub-id></citation>
</ref>
<ref id="ref49">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Laing</surname> <given-names>C. R.</given-names></name> <name><surname>Chow</surname> <given-names>C. C.</given-names></name></person-group> (<year>2002</year>). <article-title>A spiking neuron model for binocular rivalry</article-title>. <source>J. Comput. Neurosci.</source> <volume>12</volume>, <fpage>39</fpage>&#x2013;<lpage>53</lpage>. doi: <pub-id pub-id-type="doi">10.1023/A:1014942129705</pub-id></citation>
</ref>
<ref id="ref50">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lam</surname> <given-names>N. H.</given-names></name> <name><surname>Borduqui</surname> <given-names>T.</given-names></name> <name><surname>Hallak</surname> <given-names>J.</given-names></name> <name><surname>Roque</surname> <given-names>A.</given-names></name> <name><surname>Anticevic</surname> <given-names>A</given-names></name></person-group>. (<year>2022</year>). <article-title>Effects of altered excitation-inhibition balance on decision making in a cortical circuit model</article-title>. <source>J. Neurosci.</source> <volume>42</volume>, <fpage>1035</fpage>&#x2013;<lpage>1053</lpage>. doi: <pub-id pub-id-type="doi">10.1523/JNEUROSCI.1371-20.2021</pub-id>, PMID: <pub-id pub-id-type="pmid">34887320</pub-id></citation>
</ref>
<ref id="ref51">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Leopold</surname> <given-names>D. A.</given-names></name> <name><surname>Logothetis</surname> <given-names>N. K.</given-names></name></person-group> (<year>1999</year>). <article-title>Multistable phenomena: changing views in perception</article-title>. <source>Trends Cogn. Sci.</source> <volume>3</volume>, <fpage>254</fpage>&#x2013;<lpage>264</lpage>. doi: <pub-id pub-id-type="doi">10.1016/S1364-6613(99)01332-7</pub-id>, PMID: <pub-id pub-id-type="pmid">10377540</pub-id></citation>
</ref>
<ref id="ref52">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lorteije</surname> <given-names>J. A. M.</given-names></name> <name><surname>Zylberberg</surname> <given-names>A.</given-names></name> <name><surname>Ouellette</surname> <given-names>B. G.</given-names></name> <name><surname>De Zeeuw</surname> <given-names>C. I.</given-names></name> <name><surname>Sigman</surname> <given-names>M.</given-names></name> <name><surname>Roelfsema</surname> <given-names>P. R.</given-names></name></person-group> (<year>2015</year>). <article-title>The formation of hierarchical decisions in the visual cortex</article-title>. <source>Neuron</source> <volume>87</volume>, <fpage>1344</fpage>&#x2013;<lpage>1356</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.neuron.2015.08.015</pub-id>, PMID: <pub-id pub-id-type="pmid">26365766</pub-id></citation>
</ref>
<ref id="ref53">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Marcos</surname> <given-names>E.</given-names></name> <name><surname>Genovesio</surname> <given-names>A.</given-names></name></person-group> (<year>2016</year>). <article-title>Determining monkey free choice long before the choice is made: the principal role of prefrontal neurons involved in both decision and motor processes</article-title>. <source>Front. Neural Circuits</source> <volume>10</volume>:<fpage>75</fpage>. doi: <pub-id pub-id-type="doi">10.3389/FNCIR.2016.00075</pub-id></citation>
</ref>
<ref id="ref54">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Marcos</surname> <given-names>E.</given-names></name> <name><surname>Pani</surname> <given-names>P.</given-names></name> <name><surname>Brunamonti</surname> <given-names>E.</given-names></name> <name><surname>Deco</surname> <given-names>G.</given-names></name> <name><surname>Ferraina</surname> <given-names>S.</given-names></name> <name><surname>Verschure</surname> <given-names>P.</given-names></name></person-group> (<year>2013</year>). <article-title>Neural variability in premotor cortex is modulated by trial history and predicts behavioral performance</article-title>. <source>Neuron</source> <volume>78</volume>, <fpage>249</fpage>&#x2013;<lpage>255</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.neuron.2013.02.006</pub-id>, PMID: <pub-id pub-id-type="pmid">23622062</pub-id></citation>
</ref>
<ref id="ref55">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Marsaglia</surname> <given-names>G.</given-names></name> <name><surname>Tsang</surname> <given-names>W. W.</given-names></name> <name><surname>Wang</surname> <given-names>J.</given-names></name></person-group> (<year>2003</year>). <article-title>Evaluating Kolmogorov&#x2019;s distribution</article-title>. <source>J. Stat. Softw.</source> <volume>8</volume>, <fpage>1</fpage>&#x2013;<lpage>4</lpage>. doi: <pub-id pub-id-type="doi">10.18637/JSS.V008.I18</pub-id></citation>
</ref>
<ref id="ref56">
<citation citation-type="journal"><person-group person-group-type="author">
<name><surname>Minsky</surname> <given-names>M.</given-names></name>
</person-group> (<year>1961</year>). <article-title>Steps toward artificial intelligence</article-title>. <source>Proc. IRE</source> <volume>49</volume>, <fpage>8</fpage>&#x2013;<lpage>30</lpage>. doi: <pub-id pub-id-type="doi">10.1109/JRPROC.1961.287775</pub-id>, PMID: <pub-id pub-id-type="pmid">38858069</pub-id></citation>
</ref>
<ref id="ref57">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Moreno-Bote</surname> <given-names>R.</given-names></name> <name><surname>Rinzel</surname> <given-names>J.</given-names></name> <name><surname>Rubin</surname> <given-names>N.</given-names></name></person-group> (<year>2007</year>). <article-title>Noise-induced alternations in an attractor network model of perceptual bistability</article-title>. <source>J. Neurophysiol.</source> <volume>98</volume>, <fpage>1125</fpage>&#x2013;<lpage>1139</lpage>. doi: <pub-id pub-id-type="doi">10.1152/jn.00116.2007</pub-id>, PMID: <pub-id pub-id-type="pmid">17615138</pub-id></citation>
</ref>
<ref id="ref58">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Nagengast</surname> <given-names>A. J.</given-names></name> <name><surname>Braun</surname> <given-names>D. A.</given-names></name> <name><surname>Wolpert</surname> <given-names>D. M.</given-names></name></person-group> (<year>2011</year>). <article-title>Risk sensitivity in a motor task with speed-accuracy trade-off</article-title>. <source>J. Neurophysiol.</source> <volume>105</volume>, <fpage>2668</fpage>&#x2013;<lpage>2674</lpage>. doi: <pub-id pub-id-type="doi">10.1152/jn.00804.2010</pub-id>, PMID: <pub-id pub-id-type="pmid">21430284</pub-id></citation>
</ref>
<ref id="ref59">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Nikulin</surname> <given-names>M. S.</given-names></name> <name><surname>Chimitova</surname> <given-names>E. V.</given-names></name></person-group> (<year>2017</year>). <article-title>Comparison of the chi-squared goodness-of-fit test with other tests</article-title>. <source>Chi-squared Goodness-of-fit Tests for Censored Data</source>, <fpage>71</fpage>&#x2013;<lpage>86</lpage>. doi: <pub-id pub-id-type="doi">10.1002/9781119427605.CH3</pub-id></citation>
</ref>
<ref id="ref60">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>O&#x2019;Brien</surname> <given-names>M. K.</given-names></name> <name><surname>Ahmed</surname> <given-names>A. A.</given-names></name></person-group> (<year>2015</year>). <article-title>Threat affects risk preferences in movement decision making</article-title>. <source>Front. Behav. Neurosci.</source> <volume>9</volume>:<fpage>150</fpage>. doi: <pub-id pub-id-type="doi">10.3389/fnbeh.2015.00150</pub-id>, PMID: <pub-id pub-id-type="pmid">26106311</pub-id></citation>
</ref>
<ref id="ref61">
<citation citation-type="journal"><person-group person-group-type="author">
<name><surname>Padoa-Schioppa</surname> <given-names>C.</given-names></name>
</person-group> (<year>2011</year>). <article-title>Neurobiology of economic choice: A good-based model</article-title>. <source>Ann. Rev. Neurosci.</source> <volume>34</volume>, <fpage>333</fpage>&#x2013;<lpage>359</lpage>. doi: <pub-id pub-id-type="doi">10.1146/ANNUREV-NEURO-061010-113648</pub-id></citation>
</ref>
<ref id="ref62">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Park</surname> <given-names>S. Q.</given-names></name> <name><surname>Kahnt</surname> <given-names>T.</given-names></name> <name><surname>Rieskamp</surname> <given-names>J.</given-names></name> <name><surname>Heekeren</surname> <given-names>H. R.</given-names></name></person-group> (<year>2011</year>). <article-title>Neurobiology of value integration: when value impacts valuation</article-title>. <source>J. Neurosci.</source> <volume>31</volume>, <fpage>9307</fpage>&#x2013;<lpage>9314</lpage>. doi: <pub-id pub-id-type="doi">10.1523/JNEUROSCI.4973-10.2011</pub-id>, PMID: <pub-id pub-id-type="pmid">21697380</pub-id></citation>
</ref>
<ref id="ref63">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Pastor-Bernier</surname> <given-names>A.</given-names></name> <name><surname>Cisek</surname> <given-names>P.</given-names></name></person-group> (<year>2011</year>). <article-title>Neural correlates of biased competition in premotor cortex</article-title>. <source>J. Neurosci.</source> <volume>31</volume>, <fpage>7083</fpage>&#x2013;<lpage>7088</lpage>. doi: <pub-id pub-id-type="doi">10.1523/JNEUROSCI.5681-10.2011</pub-id>, PMID: <pub-id pub-id-type="pmid">21562270</pub-id></citation>
</ref>
<ref id="ref64">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Quinn</surname> <given-names>G. P.</given-names></name> <name><surname>Keough</surname> <given-names>M. J.</given-names></name></person-group> (<year>2002</year>). <article-title>Experimental design and data analysis for biologists</article-title>. <source>Exp. Design Data Analysis Biol.</source> doi: <pub-id pub-id-type="doi">10.1017/CBO9780511806384</pub-id></citation>
</ref>
<ref id="ref65">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ratcliff</surname> <given-names>R.</given-names></name> <name><surname>McKoon</surname> <given-names>G.</given-names></name></person-group> (<year>2008</year>). <article-title>The diffusion decision model: theory and data for two-choice decision tasks</article-title>. <source>Neural Comput.</source> <volume>20</volume>:<fpage>873</fpage>. doi: <pub-id pub-id-type="doi">10.1162/NECO.2008.12-06-420</pub-id></citation>
</ref>
<ref id="ref66">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Roitman</surname> <given-names>J. D.</given-names></name> <name><surname>Shadlen</surname> <given-names>M. N.</given-names></name></person-group> (<year>2002</year>). <article-title>Response of neurons in the lateral intraparietal area during a combined visual discrimination reaction time task</article-title>. <source>J. Neurosci.</source> <volume>22</volume>, <fpage>9475</fpage>&#x2013;<lpage>9489</lpage>. doi: <pub-id pub-id-type="doi">10.1523/JNEUROSCI.22-21-09475.2002</pub-id>, PMID: <pub-id pub-id-type="pmid">12417672</pub-id></citation>
</ref>
<ref id="ref67">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Roxin</surname> <given-names>A.</given-names></name> <name><surname>Ledberg</surname> <given-names>A.</given-names></name></person-group> (<year>2008</year>). <article-title>Neurobiological models of two-choice decision making can be reduced to a one-dimensional nonlinear diffusion equation</article-title>. <source>PLoS Comput. Biol.</source> <volume>4</volume>:<fpage>e1000046</fpage>. doi: <pub-id pub-id-type="doi">10.1371/journal.pcbi.1000046</pub-id>, PMID: <pub-id pub-id-type="pmid">18369436</pub-id></citation>
</ref>
<ref id="ref68">
<citation citation-type="journal"><person-group person-group-type="author">
<name><surname>Rubin</surname> <given-names>N.</given-names></name>
</person-group> (<year>2003</year>). <article-title>Binocular rivalry and perceptual multi-stability</article-title>. <source>Trends Neurosci.</source> <volume>26</volume>, <fpage>289</fpage>&#x2013;<lpage>291</lpage>. doi: <pub-id pub-id-type="doi">10.1016/S0166-2236(03)00128-0</pub-id>, PMID: <pub-id pub-id-type="pmid">12798596</pub-id></citation>
</ref>
<ref id="ref69">
<citation citation-type="journal"><person-group person-group-type="author">
<name><surname>Salinas</surname> <given-names>E.</given-names></name>
</person-group> (<year>2008</year>). <article-title>So many choices: what computational models reveal about decision-making mechanisms</article-title>. <source>Neuron</source> <volume>60</volume>, <fpage>946</fpage>&#x2013;<lpage>949</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.neuron.2008.12.011</pub-id>, PMID: <pub-id pub-id-type="pmid">19109902</pub-id></citation>
</ref>
<ref id="ref70">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Schuck-Paim</surname> <given-names>C.</given-names></name> <name><surname>Kacelnik</surname> <given-names>A.</given-names></name></person-group> (<year>2007</year>). <article-title>Choice processes in multialternative decision making</article-title>. <source>Behav. Ecol.</source> <volume>18</volume>, <fpage>541</fpage>&#x2013;<lpage>550</lpage>. doi: <pub-id pub-id-type="doi">10.1093/beheco/arm005</pub-id>, PMID: <pub-id pub-id-type="pmid">37269710</pub-id></citation>
</ref>
<ref id="ref71">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Shadlen</surname> <given-names>M. N.</given-names></name> <name><surname>Newsome</surname> <given-names>W. T.</given-names></name></person-group> (<year>1996</year>). <article-title>Motion perception: seeing and deciding</article-title>. <source>Proc. Natl. Acad. Sci. USA</source> <volume>93</volume>, <fpage>628</fpage>&#x2013;<lpage>633</lpage>. doi: <pub-id pub-id-type="doi">10.1073/pnas.93.2.628</pub-id>, PMID: <pub-id pub-id-type="pmid">8570606</pub-id></citation>
</ref>
<ref id="ref72">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Shadlen</surname> <given-names>M. N.</given-names></name> <name><surname>Newsome</surname> <given-names>W. T.</given-names></name></person-group> (<year>2001</year>). <article-title>Neural basis of a perceptual decision in the parietal cortex (area LIP) of the rhesus monkey</article-title>. <source>J. Neurophysiol.</source> <volume>86</volume>, <fpage>1916</fpage>&#x2013;<lpage>1936</lpage>. doi: <pub-id pub-id-type="doi">10.1152/jn.2001.86.4.1916</pub-id>, PMID: <pub-id pub-id-type="pmid">11600651</pub-id></citation>
</ref>
<ref id="ref73">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Shahar</surname> <given-names>N.</given-names></name> <name><surname>Hauser</surname> <given-names>T. U.</given-names></name> <name><surname>Moutoussis</surname> <given-names>M.</given-names></name> <name><surname>Moran</surname> <given-names>R.</given-names></name> <name><surname>Keramati</surname> <given-names>M.</given-names></name><collab>NSPN consortium</collab> <etal/></person-group>. (<year>2019</year>). <article-title>Improving the reliability of model-based decision-making estimates in the two-stage decision task with reaction-times and drift-diffusion modeling</article-title>. <source>PLoS Comput. Biol.</source> <volume>15</volume>:<fpage>e1006803</fpage>. doi: <pub-id pub-id-type="doi">10.1371/journal.pcbi.1006803</pub-id>, PMID: <pub-id pub-id-type="pmid">30759077</pub-id></citation>
</ref>
<ref id="ref74">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Siegel</surname> <given-names>M.</given-names></name> <name><surname>Warden</surname> <given-names>M. R.</given-names></name> <name><surname>Miller</surname> <given-names>E. K.</given-names></name></person-group> (<year>2009</year>). <article-title>Phase-dependent neuronal coding of objects in short-term memory</article-title>. <source>Proc. Natl. Acad. Sci. USA</source> <volume>106</volume>, <fpage>21341</fpage>&#x2013;<lpage>21346</lpage>. doi: <pub-id pub-id-type="doi">10.1073/pnas.0908193106</pub-id>, PMID: <pub-id pub-id-type="pmid">19926847</pub-id></citation>
</ref>
<ref id="ref75">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Skvortsova</surname> <given-names>V.</given-names></name> <name><surname>Palminteri</surname> <given-names>S.</given-names></name> <name><surname>Pessiglione</surname> <given-names>M.</given-names></name></person-group> (<year>2014</year>). <article-title>Learning to minimize efforts versus maximizing rewards: computational principles and neural correlates</article-title>. <source>J. Neurosci.</source> <volume>34</volume>, <fpage>15621</fpage>&#x2013;<lpage>15630</lpage>. doi: <pub-id pub-id-type="doi">10.1523/JNEUROSCI.1350-14.2014</pub-id>, PMID: <pub-id pub-id-type="pmid">25411490</pub-id></citation>
</ref>
<ref id="ref76">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Smallwood</surname> <given-names>R. D.</given-names></name> <name><surname>Sondik</surname> <given-names>E. J.</given-names></name></person-group> (<year>1973</year>). <article-title>The optimal control of partially observable Markov processes over a finite horizon</article-title>. <source>Oper. Res.</source> <volume>21</volume>, <fpage>1071</fpage>&#x2013;<lpage>1088</lpage>. doi: <pub-id pub-id-type="doi">10.1287/OPRE.21.5.1071</pub-id></citation>
</ref>
<ref id="ref77">
<citation citation-type="journal"><person-group person-group-type="author">
<name><surname>Smirnov</surname> <given-names>N.</given-names></name>
</person-group> (<year>1948</year>). <article-title>Table for estimating the goodness of fit of empirical distributions</article-title>. <source>Annals Mathemat. Stat.</source> <volume>19</volume>, <fpage>279</fpage>&#x2013;<lpage>281</lpage>. doi: <pub-id pub-id-type="doi">10.1214/AOMS/1177730256</pub-id></citation>
</ref>
<ref id="ref78">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Soltani</surname> <given-names>A.</given-names></name> <name><surname>Lee</surname> <given-names>D.</given-names></name> <name><surname>Wang</surname> <given-names>X.-J.</given-names></name></person-group> (<year>2006</year>). <article-title>Neural mechanism for stochastic behaviour during a competitive game</article-title>. <source>Neural Netw.</source> <volume>19</volume>, <fpage>1075</fpage>&#x2013;<lpage>1090</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.neunet.2006.05.044</pub-id>, PMID: <pub-id pub-id-type="pmid">17015181</pub-id></citation>
</ref>
<ref id="ref79">
<citation citation-type="journal"><person-group person-group-type="author">
<name><surname>Stephens</surname> <given-names>M. A.</given-names></name>
</person-group> (<year>1974</year>). <article-title>EDF statistics for goodness of fit and some comparisons</article-title>. <source>J. Am. Stat. Assoc.</source> <volume>69</volume>, <fpage>730</fpage>&#x2013;<lpage>737</lpage>. doi: <pub-id pub-id-type="doi">10.1080/01621459.1974.10480196</pub-id></citation>
</ref>
<ref id="ref80">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sutton</surname> <given-names>R. S.</given-names></name> <name><surname>Barto</surname> <given-names>A. G.</given-names></name></person-group> (<year>1981</year>). <article-title>Toward a modern theory of adaptive networks: expectation and prediction</article-title>. <source>Psychol. Rev.</source> <volume>88</volume>, <fpage>135</fpage>&#x2013;<lpage>170</lpage>. doi: <pub-id pub-id-type="doi">10.1037/0033-295X.88.2.135</pub-id>, PMID: <pub-id pub-id-type="pmid">7291377</pub-id></citation>
</ref>
<ref id="ref81">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Sutton</surname> <given-names>R. S.</given-names></name> <name><surname>Barto</surname> <given-names>A. G.</given-names></name></person-group> (<year>2018</year>). <source><italic>Reinforcement learning</italic>, second</source>. <publisher-loc>American Psychological Association</publisher-loc>: <publisher-name>MIT Press</publisher-name>.</citation>
</ref>
<ref id="ref83">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Thura</surname> <given-names>D.</given-names></name> <name><surname>Cisek</surname> <given-names>P.</given-names></name></person-group> (<year>2016</year>). <article-title>Modulation of premotor and primary motor cortical activity during volitional adjustments of speed-accuracy trade-offs</article-title>. <source>J. Neurosci.</source> <volume>36</volume>, <fpage>938</fpage>&#x2013;<lpage>956</lpage>. doi: <pub-id pub-id-type="doi">10.1523/JNEUROSCI.2230-15.2016</pub-id>, PMID: <pub-id pub-id-type="pmid">26791222</pub-id></citation>
</ref>
<ref id="ref82">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Thura</surname> <given-names>D.</given-names></name> <name><surname>Cabana</surname> <given-names>J. F.</given-names></name> <name><surname>Feghaly</surname> <given-names>A.</given-names></name> <name><surname>Cisek</surname> <given-names>P.</given-names></name></person-group> (<year>2022</year>). <article-title>Integrated neural dynamics of sensorimotor decisions and actions</article-title>. <source>PLOS Biology</source>. <volume>20</volume>:<fpage>e3001861</fpage>. doi: <pub-id pub-id-type="doi">10.1371/journal.pbio.3001861</pub-id>, PMID: <pub-id pub-id-type="pmid">26791222</pub-id></citation>
</ref>
<ref id="ref84">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Trommersh&#x00E4;user</surname> <given-names>J.</given-names></name> <name><surname>Maloney</surname> <given-names>L. T.</given-names></name> <name><surname>Landy</surname> <given-names>M. S.</given-names></name></person-group> (<year>2008</year>). <article-title>Decision making, movement planning and statistical decision theory</article-title>. <source>Trends Cogn. Sci.</source> <volume>12</volume>, <fpage>291</fpage>&#x2013;<lpage>297</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.tics.2008.04.010</pub-id>, PMID: <pub-id pub-id-type="pmid">18614390</pub-id></citation>
</ref>
<ref id="ref85">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Verbeke</surname> <given-names>G.</given-names></name> <name><surname>Molenberghs</surname> <given-names>G.</given-names></name></person-group> (<year>2009</year>). <source><italic>Linear mixed models for longitudinal data</italic>. In springer series in statistics</source>. <publisher-loc>New York</publisher-loc>: <publisher-name>Springer</publisher-name>.</citation>
</ref>
<ref id="ref86">
<citation citation-type="journal"><person-group person-group-type="author">
<name><surname>Wallis</surname> <given-names>J. D.</given-names></name>
</person-group> (<year>2011</year>). <article-title>Cross-species studies of orbitofrontal cortex and value-based decision-making</article-title>. <source>Nat. Neurosci.</source> <volume>15</volume>, <fpage>13</fpage>&#x2013;<lpage>19</lpage>. doi: <pub-id pub-id-type="doi">10.1038/nn.2956</pub-id>, PMID: <pub-id pub-id-type="pmid">22101646</pub-id></citation>
</ref>
<ref id="ref87">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wallis</surname> <given-names>J. D.</given-names></name> <name><surname>Kennerley</surname> <given-names>S. W.</given-names></name></person-group> (<year>2011</year>). <article-title>Contrasting reward signals in the orbitofrontal cortex and anterior cingulate cortex</article-title>. <source>Ann. N. Y. Acad. Sci.</source> <volume>1239</volume>, <fpage>33</fpage>&#x2013;<lpage>42</lpage>. doi: <pub-id pub-id-type="doi">10.1111/J.1749-6632.2011.06277.X</pub-id></citation>
</ref>
<ref id="ref88">
<citation citation-type="journal"><person-group person-group-type="author">
<name><surname>Wang</surname> <given-names>X. J.</given-names></name>
</person-group> (<year>2002</year>). <article-title>Probabilistic decision making by slow reverberation in cortical circuits</article-title>. <source>Neuron</source> <volume>36</volume>, <fpage>955</fpage>&#x2013;<lpage>968</lpage>. doi: <pub-id pub-id-type="doi">10.1016/S0896-6273(02)01092-9</pub-id>, PMID: <pub-id pub-id-type="pmid">12467598</pub-id></citation>
</ref>
<ref id="ref89">
<citation citation-type="journal"><person-group person-group-type="author">
<name><surname>Wang</surname> <given-names>X. J.</given-names></name>
</person-group> (<year>2008</year>). <article-title>Decision making in recurrent neuronal circuits</article-title>. <source>Neuron</source> <volume>60</volume>, <fpage>215</fpage>&#x2013;<lpage>234</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.neuron.2008.09.034</pub-id>, PMID: <pub-id pub-id-type="pmid">18957215</pub-id></citation>
</ref>
<ref id="ref90">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Webb</surname> <given-names>T. J.</given-names></name> <name><surname>Rolls</surname> <given-names>E. T.</given-names></name> <name><surname>Deco</surname> <given-names>G.</given-names></name> <name><surname>Feng</surname> <given-names>J.</given-names></name></person-group> (<year>2011</year>). <article-title>Noise in attractor networks in the brain produced by graded firing rate representations</article-title>. <source>PLoS One</source> <volume>6</volume>:<fpage>e23630</fpage>. doi: <pub-id pub-id-type="doi">10.1371/journal.pone.0023630</pub-id>, PMID: <pub-id pub-id-type="pmid">21931607</pub-id></citation>
</ref>
<ref id="ref91">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wessel</surname> <given-names>J. R.</given-names></name> <name><surname>Aron</surname> <given-names>A. R.</given-names></name></person-group> (<year>2017</year>). <article-title>On the Globality of motor suppression: unexpected events and their influence on behavior and cognition</article-title>. <source>Neuron</source> <volume>93</volume>, <fpage>259</fpage>&#x2013;<lpage>280</lpage>. doi: <pub-id pub-id-type="doi">10.1016/j.neuron.2016.12.013</pub-id>, PMID: <pub-id pub-id-type="pmid">28103476</pub-id></citation>
</ref>
<ref id="ref92">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>White</surname> <given-names>C. N.</given-names></name> <name><surname>Servant</surname> <given-names>M.</given-names></name> <name><surname>Logan</surname> <given-names>G. D.</given-names></name></person-group> (<year>2018</year>). <article-title>Testing the validity of conflict drift-diffusion models for use in estimating cognitive processes: a parameter-recovery study</article-title>. <source>Psychon. Bull. Rev.</source> <volume>25</volume>, <fpage>286</fpage>&#x2013;<lpage>301</lpage>. doi: <pub-id pub-id-type="doi">10.3758/s13423-017-1271-2</pub-id>, PMID: <pub-id pub-id-type="pmid">28357629</pub-id></citation>
</ref>
<ref id="ref93">
<citation citation-type="journal"><person-group person-group-type="author">
<name><surname>Wilson</surname> <given-names>H. R.</given-names></name>
</person-group> (<year>2003</year>). <article-title>Computational evidence for a rivalry hierarchy in vision</article-title>. <source>Proc. Natl. Acad. Sci. USA</source> <volume>100</volume>, <fpage>14499</fpage>&#x2013;<lpage>14503</lpage>. doi: <pub-id pub-id-type="doi">10.1073/PNAS.2333622100/ASSET/A7DC6A54-0867-4BFF-B228-7E54DCDDB0A3/ASSETS/GRAPHIC/PQ2333622005.JPEG</pub-id></citation>
</ref>
<ref id="ref94">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wilson</surname> <given-names>H. R.</given-names></name> <name><surname>Cowan</surname> <given-names>J. D.</given-names></name></person-group> (<year>1972</year>). <article-title>Excitatory and inhibitory interactions in localized populations of model neurons</article-title>. <source>Biophys. J.</source> <volume>12</volume>, <fpage>1</fpage>&#x2013;<lpage>24</lpage>. doi: <pub-id pub-id-type="doi">10.1016/S0006-3495(72)86068-5</pub-id>, PMID: <pub-id pub-id-type="pmid">4332108</pub-id></citation>
</ref>
<ref id="ref95">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wong</surname> <given-names>K. F.</given-names></name> <name><surname>Huk</surname> <given-names>A. C.</given-names></name> <name><surname>Shadlen</surname> <given-names>M. N.</given-names></name> <name><surname>Wang</surname> <given-names>X. J.</given-names></name></person-group> (<year>2007</year>). <article-title>Neural circuit dynamics underlying accumulation of time-varying evidence during perceptual decision making</article-title>. <source>Front. Comput. Neurosci.</source> <volume>1</volume>:<fpage>6</fpage>. doi: <pub-id pub-id-type="doi">10.3389/neuro.10.006.2007</pub-id>, PMID: <pub-id pub-id-type="pmid">18946528</pub-id></citation>
</ref>
<ref id="ref96">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wong</surname> <given-names>K. F.</given-names></name> <name><surname>Wang</surname> <given-names>X. J.</given-names></name></person-group> (<year>2006</year>). <article-title>A recurrent network mechanism of time integration in perceptual decisions</article-title>. <source>J. Neurosci.</source> <volume>26</volume>, <fpage>1314</fpage>&#x2013;<lpage>1328</lpage>. doi: <pub-id pub-id-type="doi">10.1523/JNEUROSCI.3733-05.2006</pub-id>, PMID: <pub-id pub-id-type="pmid">16436619</pub-id></citation>
</ref>
<ref id="ref97">
<citation citation-type="journal"><person-group person-group-type="author">
<name><surname>Zylberberg</surname> <given-names>A.</given-names></name>
</person-group> (<year>2022</year>). <article-title>Decision prioritization and causal reasoning in decision hierarchies</article-title>. <source>PLoS Comput. Biol.</source> <volume>17</volume>, <fpage>1</fpage>&#x2013;<lpage>39</lpage>. doi: <pub-id pub-id-type="doi">10.1371/journal.pcbi.1009688</pub-id></citation>
</ref>
<ref id="ref98">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zylberberg</surname> <given-names>A.</given-names></name> <name><surname>Lorteije</surname> <given-names>J. A. M.</given-names></name> <name><surname>Ouellette</surname> <given-names>B. G.</given-names></name> <name><surname>De Zeeuw</surname> <given-names>C. I.</given-names></name> <name><surname>Sigman</surname> <given-names>M.</given-names></name> <name><surname>Roelfsema</surname> <given-names>P.</given-names></name></person-group> (<year>2017</year>). <article-title>Serial, parallel and hierarchical decision making in primates</article-title>. <source>eLife</source> <volume>6</volume>:<fpage>e17331</fpage>. doi: <pub-id pub-id-type="doi">10.7554/eLife.17331</pub-id>, PMID: <pub-id pub-id-type="pmid">28648172</pub-id></citation>
</ref>
</ref-list>
</back>
</article>