<?xml version="1.0" encoding="UTF-8" standalone="no"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.3 20070202//EN" "journalpublishing.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article">
<front>
<journal-meta>
<journal-id journal-id-type="publisher-id">Front. Neurosci.</journal-id>
<journal-title>Frontiers in Neuroscience</journal-title>
<abbrev-journal-title abbrev-type="pubmed">Front. Neurosci.</abbrev-journal-title>
<issn pub-type="epub">1662-453X</issn>
<publisher>
<publisher-name>Frontiers Media S.A.</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="doi">10.3389/fnins.2017.00598</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Neuroscience</subject>
<subj-group>
<subject>Original Research</subject>
</subj-group>
</subj-group>
</article-categories>
<title-group>
<article-title>Cardiac Concomitants of Feedback and Prediction Error Processing in Reinforcement Learning</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name><surname>Kastner</surname> <given-names>Lucas</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="author-notes" rid="fn003"><sup>&#x02020;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/402566/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Kube</surname> <given-names>Jana</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="aff" rid="aff3"><sup>3</sup></xref>
<xref ref-type="author-notes" rid="fn003"><sup>&#x02020;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/406944/overview"/>
</contrib>
<contrib contrib-type="author">
<name><surname>Villringer</surname> <given-names>Arno</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="aff" rid="aff4"><sup>4</sup></xref>
<xref ref-type="aff" rid="aff5"><sup>5</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/2147/overview"/>
</contrib>
<contrib contrib-type="author" corresp="yes">
<name><surname>Neumann</surname> <given-names>Jane</given-names></name>
<xref ref-type="aff" rid="aff1"><sup>1</sup></xref>
<xref ref-type="aff" rid="aff2"><sup>2</sup></xref>
<xref ref-type="aff" rid="aff6"><sup>6</sup></xref>
<xref ref-type="author-notes" rid="fn001"><sup>&#x0002A;</sup></xref>
<uri xlink:href="http://loop.frontiersin.org/people/8954/overview"/>
</contrib>
</contrib-group>
<aff id="aff1"><sup>1</sup><institution>IFB Adiposity Diseases, Leipzig University Medical Center</institution>, <addr-line>Leipzig</addr-line>, <country>Germany</country></aff>
<aff id="aff2"><sup>2</sup><institution>Max Planck Institute for Human Cognitive and Brain Sciences</institution>, <addr-line>Leipzig</addr-line>, <country>Germany</country></aff>
<aff id="aff3"><sup>3</sup><institution>Faculty 5&#x02013;Business, Law and Social Sciences, Brandenburg University of Technology Cottbus&#x02013;Senftenberg</institution>, <addr-line>Cottbus</addr-line>, <country>Germany</country></aff>
<aff id="aff4"><sup>4</sup><institution>Clinic of Cognitive Neurology, University Hospital Leipzig</institution>, <addr-line>Leipzig</addr-line>, <country>Germany</country></aff>
<aff id="aff5"><sup>5</sup><institution>Mind and Brain Institute, Berlin School of Mind and Brain, Humboldt-University</institution>, <addr-line>Berlin</addr-line>, <country>Germany</country></aff>
<aff id="aff6"><sup>6</sup><institution>Department of Medical Engineering and Biotechnology, University of Applied Sciences</institution>, <addr-line>Jena</addr-line>, <country>Germany</country></aff>
<author-notes>
<fn fn-type="edited-by"><p>Edited by: Yoko Nagai, Brighton and Sussex Medical School, United Kingdom</p></fn>
<fn fn-type="edited-by"><p>Reviewed by: Karl-J&#x000FC;rgen B&#x000E4;r, Friedrich-Schiller-Universit&#x000E4;t Jena, Germany; Chris Mark Fiacconi, University of Western Ontario, Canada</p></fn>
<fn fn-type="corresp" id="fn001"><p>&#x0002A;Correspondence: Jane Neumann <email>neumann&#x00040;cbs.mpg.de</email></p></fn>
<fn fn-type="other" id="fn002"><p>This article was submitted to Autonomic Neuroscience, a section of the journal Frontiers in Neuroscience</p></fn>
<fn fn-type="other" id="fn003"><p>&#x02020;These authors have contributed equally to this work.</p></fn></author-notes>
<pub-date pub-type="epub">
<day>30</day>
<month>10</month>
<year>2017</year>
</pub-date>
<pub-date pub-type="collection">
<year>2017</year>
</pub-date>
<volume>11</volume>
<elocation-id>598</elocation-id>
<history>
<date date-type="received">
<day>07</day>
<month>03</month>
<year>2017</year>
</date>
<date date-type="accepted">
<day>11</day>
<month>10</month>
<year>2017</year>
</date>
</history>
<permissions>
<copyright-statement>Copyright &#x000A9; 2017 Kastner, Kube, Villringer and Neumann.</copyright-statement>
<copyright-year>2017</copyright-year>
<copyright-holder>Kastner, Kube, Villringer and Neumann</copyright-holder>
<license xlink:href="http://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (CC BY). The use, distribution or reproduction in other forums is permitted, provided the original author(s) or licensor are credited and that the original publication in this journal is cited, in accordance with accepted academic practice. No use, distribution or reproduction is permitted which does not comply with these terms.</p></license>
</permissions>
<abstract><p>Successful learning hinges on the evaluation of positive and negative feedback. We assessed differential learning from reward and punishment in a monetary reinforcement learning paradigm, together with cardiac concomitants of positive and negative feedback processing. On the behavioral level, learning from reward resulted in more advantageous behavior than learning from punishment, suggesting a differential impact of reward and punishment on successful feedback-based learning. On the autonomic level, learning and feedback processing were closely mirrored by phasic cardiac responses on a trial-by-trial basis: (1) Negative feedback was accompanied by faster and prolonged heart rate deceleration compared to positive feedback. (2) Cardiac responses shifted from feedback presentation at the beginning of learning to stimulus presentation later on. (3) Most importantly, the strength of phasic cardiac responses to the presentation of feedback correlated with the strength of prediction error signals that alert the learner to the necessity for behavioral adaptation. Considering participants&#x00027; weight status and gender revealed obesity-related deficits in learning to avoid negative consequences and less consistent behavioral adaptation in women compared to men. In sum, our results provide strong new evidence for the notion that during learning phasic cardiac responses reflect an internal value and feedback monitoring system that is sensitive to the violation of performance-based expectations. Moreover, inter-individual differences in weight status and gender may affect both behavioral and autonomic responses in reinforcement-based learning.</p></abstract>
<kwd-group>
<kwd>reinforcement learning</kwd>
<kwd>prediction error</kwd>
<kwd>reward</kwd>
<kwd>punishment</kwd>
<kwd>heart rate</kwd>
<kwd>gender</kwd>
<kwd>obesity</kwd>
</kwd-group>
<contract-num rid="cn001">FKZ: 01EO1001</contract-num>
<contract-num rid="cn002">SFB 1052-A5</contract-num>
<contract-sponsor id="cn001">Bundesministerium f&#x000FC;r Bildung und Forschung<named-content content-type="fundref-id">10.13039/501100002347</named-content></contract-sponsor>
<contract-sponsor id="cn002">Deutsche Forschungsgemeinschaft<named-content content-type="fundref-id">10.13039/501100001659</named-content></contract-sponsor>
<counts>
<fig-count count="5"/>
<table-count count="1"/>
<equation-count count="6"/>
<ref-count count="105"/>
<page-count count="19"/>
<word-count count="16474"/>
</counts>
</article-meta>
</front>
<body>
<sec sec-type="intro" id="s1">
<title>Introduction</title>
<p>Reinforcement learning describes the process of adapting behavior according to the consequences of actions. Actions or choices that lead to reward or positive feedback should be repeated in similar future situations, whereas actions or choices followed by punishment or negative feedback should be avoided. Thus, in reinforcement learning positive and negative feedback provide the learner with the necessary information for successful behavioral adaptation.</p>
<p>In this paper, we wished to address three research questions: (1) Can we observe systematic differences in leaning from reward and learning from punishment in reinforcement learning? (2) How is feedback processing and learning reflected in phasic cardiac responses during reinforcement learning? (3) How do gender and weight status impact on behavioral measures and cardiac concomitants of reinforcement learning? Thus, we applied an experimental design that comprised independent reward and punishment conditions. During task performance continuous ECG measurements were obtained. Moreover, computational modeling was applied to behavioral and autonomic measures. The paper is structured as follows. We first introduce the main concepts of reinforcement learning and phasic cardiac responses and derive hypotheses for our research questions, followed by the presentation of our experimental task, measurement techniques and analysis methods. We then present our results in the order of our research question, i.e., first regarding the behavioral level, second regarding the autonomic level, and finally regarding the effects of weight status and gender on both behavioral and autonomic measures.</p>
<p>Reinforcement learning in humans has been widely studied in health and disease, and impaired reinforcement learning mechanisms have been identified in various psychiatric and neurological disorders including Parkinson&#x00027;s disease, Huntington&#x00027;s disease, depression, schizophrenia, and several addictive disorders (e.g., de Ruiter et al., <xref ref-type="bibr" rid="B20">2009</xref>; Park et al., <xref ref-type="bibr" rid="B71">2010</xref>; Gradin et al., <xref ref-type="bibr" rid="B37">2011</xref>; Maia and Frank, <xref ref-type="bibr" rid="B59">2011</xref>). Some studies thereby point at differential impairments in learning from reward and learning from punishment (e.g., Frank et al., <xref ref-type="bibr" rid="B28">2004</xref>; Mathar et al., <xref ref-type="bibr" rid="B64">2017b</xref>). In healthy populations, several studies highlight parallels in learning from reward and punishment (Kim et al., <xref ref-type="bibr" rid="B51">2006</xref>; Delgado et al., <xref ref-type="bibr" rid="B22">2008</xref>) including the critical involvement of the brain&#x00027;s dopaminergic system in both learning mechanisms (Glimcher, <xref ref-type="bibr" rid="B36">2011</xref>; Mathar et al., <xref ref-type="bibr" rid="B64">2017b</xref>). However, previous research also identified differences such as, increased reaction times in punishment&#x02014;compared to reward-based learning, a tendency for reduced learning from punishment, and differential functional brain responses in relation to reward and punishment, sometimes even in the absence of detectable differences in task performance (Delgado et al., <xref ref-type="bibr" rid="B23">2000</xref>; Robinson et al., <xref ref-type="bibr" rid="B76">2010a</xref>; Mattfeld et al., <xref ref-type="bibr" rid="B65">2011</xref>). The involvement of partially different neutrotransmitter systems in reward and punishment processing provides additional evidence for distinct albeit overlapping processing mechanisms for reward and punishment (Guitart-Masip et al., <xref ref-type="bibr" rid="B40">2014</xref>; Jocham et al., <xref ref-type="bibr" rid="B48">2014</xref>). Thus, processing of reward and punishment has to be considered differentially in the investigation of feedback-based learning.</p>
<p>The first goal of our study was a systematic assessment of potential differences in learning from reward and learning from punishment. We employed a probabilistic reinforcement learning paradigm consisting of independent reward and punishment conditions, where learners were provided with only positive and only negative feedback, respectively. Ecological validity of the task and participants&#x00027; task comprehension were tested by valence and arousal ratings for the presented stimuli prior to and after learning. The overall score achieved at the end of the experiment and the number of participants&#x00027; advantageous choices and response times were examined as measures of task performance. Participants&#x00027; choice inconsistency was assessed as measures for behavioral adaptation. Behavioral assessment was complemented by computational modeling that facilitates a deep and detailed analysis of learning on a trial-by-trial basis.</p>
<p>In reinforcement learning, behavioral adaptations are driven by the prediction error (PE) signal (Sch&#x000F6;nberg et al., <xref ref-type="bibr" rid="B80">2007</xref>). The PE signal encodes the deviations between the expected and the actual outcome of an action. A positive PE arises in situations where an outcome is better than expected, and a negative PE signifies that an outcome is worse. The strength or amplitude of the PE reflects the degree of deviation between expected and actual outcome, whereby fully unexpected and surprising events result in larger PEs. Strength and directionality of the PE signal determine how much and in which direction our current behavior should be adapted for the future. On the neural level, the PE signal is encoded in dopaminergic structures of the midbrain and relayed from there to striatal and prefrontal target regions to drive learning (Schultz et al., <xref ref-type="bibr" rid="B84">1997</xref>; Schultz, <xref ref-type="bibr" rid="B83">2002</xref>).</p>
<p>The construction of PE signals during learning relies on multiple skills starting with the ability to constantly monitor incoming feedback and to correctly build and maintain value representations. Further, value representations have to be updated over the course of learning, and behavior has to be adjusted accordingly for future actions and decisions. Based on behavioral observations alone, these various aspects of the learning process cannot clearly be disentangled. Computational neuroscience provides established mathematical models for reinforcement learning that implement the different aspects of learning and thus facilitate their detailed analysis. When applied to individual behavioral data, these models identify inter-individual differences in learning performance and decision strategies (e.g., Rodriguez et al., <xref ref-type="bibr" rid="B78">2006</xref>; Klein et al., <xref ref-type="bibr" rid="B52">2007</xref>; Lee et al., <xref ref-type="bibr" rid="B57">2014</xref>; Mathar et al., <xref ref-type="bibr" rid="B64">2017b</xref>) and facilitate the estimation of trials-wise PE signals and subject-specific model parameters from the data (Sutton and Barto, <xref ref-type="bibr" rid="B89">1998</xref>; Gl&#x000E4;scher and O&#x00027;Doherty, <xref ref-type="bibr" rid="B35">2010</xref>). Commonly, reinforcement learning models estimate in each trial value representations for the available options. Once an option was chosen by the learner, the PE signal is calculated as the difference between the corresponding value representation and the observed feedback. Value representations are then updated according to the strength and directionality of the PE signal. The most important model parameter in computational reinforcement learning models is the learning rate &#x003B1;. This model parameter is specific for each participant and determines the degree to which value representations are updated after feedback. In other words, the learning rate reflects how strongly new experiences in one trial impact on the participant&#x00027;s knowledge acquired over all previous trials. In addition, most models, including ours, provide a consistency parameter &#x003B2; which reflects how deterministic or stochastic the learner behaves over the course of the experiment. Comparing this parameter to the observed switching behavior of the participant provides a measure for model adequacy, i.e., for how well the computational model captures participants&#x00027; behavior.</p>
<p>Complementing our behavioral analysis, we fitted for each participant a computational reinforcement learning model to the behavioral data. The consistency parameter was used to ensure model adequacy. Subsequently, trial-wise PE signals and subject-specific learning rates were derived from the model and analyzed to identify the sources of observed behavioral effects.</p>
<p>For our behavioral and computational modeling analysis we derived the following two hypotheses from previous research: <bold>Hypothesis 1:</bold> Across participants, reinforcement learning should be reflected by an increase in correct responses and overall task score as well as a decrease in reaction times over the course of the experiment. <bold>Hypothesis 2:</bold> Learning performance and reaction times were expected to differ between reward and punishment conditions with a potentially reduced performance, smaller model-derived learning rates, and increased reaction times in the punishment condition.</p>
<p>Reactions of the autonomic nervous system provide a further invaluable source of information for the investigation of feedback processing and learning. Numerous previous studies investigated cardiac responses to external stimuli and feedback, taking into account their valence and information content. Concurrently, these studies observed a distinct pattern of phasic heart rate (HR) responses to the presentation of external stimuli: an initial cardiac deceleration that peaks within one second after stimulus onset which is followed by an acceleratory recovery to baseline after 2&#x02013;4 s. Thereby, it was consistently observed that HR deceleration is prolonged, if the presented stimulus provides negative feedback on a choice or action, in contrast to positive feedback that elicits faster acceleratory recovery (e.g., Crone et al., <xref ref-type="bibr" rid="B19">2003</xref>; Van Der Veen et al., <xref ref-type="bibr" rid="B93">2004</xref>; Groen et al., <xref ref-type="bibr" rid="B39">2007</xref>).</p>
<p>Regarding the information content of a stimulus, experimental results are less conclusive. Van Der Veen et al. (<xref ref-type="bibr" rid="B93">2004</xref>) reported prolonged HR deceleration in response to negative feedback which did not discriminate between situations where the feedback was informative or non-informative for the participant. In contrast, Mies et al. (<xref ref-type="bibr" rid="B66">2011</xref>) found transient cardiac slowing after negative feedback only in situations where the feedback was valid. In a similar vain, Groen et al. (<xref ref-type="bibr" rid="B39">2007</xref>) observed a strong deceleration in response to negative feedback that was prolonged in informative compared to non-informative feedback trials in a probabilistic learning task in children. Importantly, Groen and colleagues also reported a general reduction in feedback-related HR deceleration over the course of learning, and Crone et al. (<xref ref-type="bibr" rid="B18">2004b</xref>, <xref ref-type="bibr" rid="B16">2005</xref>) observed HR slowing already in anticipation of feedback, in particular when potentially high gains or losses were to be expected.</p>
<p>Taken together these previous studies suggested that HR deceleration in response to feedback might be caused by a deviation between an expected and an actual outcome of an action (Somsen et al., <xref ref-type="bibr" rid="B86">2000</xref>; Crone et al., <xref ref-type="bibr" rid="B19">2003</xref>). Further, they point at a shift from reliance on external feedback to an internal feedback monitoring system over the course of learning (Crone et al., <xref ref-type="bibr" rid="B18">2004b</xref>; Groen et al., <xref ref-type="bibr" rid="B39">2007</xref>). However, one major caveat of the previous work is that it provides only indirect evidence for these hypotheses, as deviations from performance-based expectations could not be assessed on a trial-by-trials basis. The second goal of our study was to provide direct evidence for a link between trial-wise performance and phasic cardiac responses during learning and feedback processing. Specifically, the use of the computational model enabled us to directly correlate the strength of PE signals with the strength of autonomic responses.</p>
<p>From the presented previous observations, we derived the following additional hypotheses for our experiment: <bold>Hypothesis 3:</bold> We expected a significant HR deceleration in response to feedback presentation which is more pronounced for negative than for positive feedback. <bold>Hypothesis 4:</bold> Over the course of learning, HR responses should shift from the presentation of feedback toward the anticipation of potential feedback already at the time of stimulus presentation. <bold>Hypothesis 5:</bold> The strength of phasic HR responses should directly predict the strength of PE signals on a trial-by-trial basis.</p>
<p>In our previous research, we identified weight status and gender as important interacting factors influencing feedback processing and reinforcement learning on the behavioral and neural level (e.g., Horstmann et al., <xref ref-type="bibr" rid="B44">2011</xref>; Garc&#x000ED;a-Garc&#x000ED;a et al., <xref ref-type="bibr" rid="B31">2014</xref>; Mathar et al., <xref ref-type="bibr" rid="B63">2017a</xref>; Kube et al., submitted). In the context of obesity, this might be explained by profound alterations of the brain&#x00027;s dopaminergic system (Wang et al., <xref ref-type="bibr" rid="B96">2001</xref>; de Weijer et al., <xref ref-type="bibr" rid="B21">2011</xref>; Volkow et al., <xref ref-type="bibr" rid="B95">2011</xref>; Horstmann et al., <xref ref-type="bibr" rid="B46">2015b</xref>) which underlies the coding of PE signals. This goes along with wide-spread obesity-related changes in both brain structure and function which extend from striatal regions into sensory and cognitive control-related frontal cortices implied in outcome processing and reinforcement learning (Horstmann et al., <xref ref-type="bibr" rid="B44">2011</xref>; Kullmann et al., <xref ref-type="bibr" rid="B55">2011</xref>; Garc&#x000ED;a-Garc&#x000ED;a et al., <xref ref-type="bibr" rid="B32">2015</xref>; Figley et al., <xref ref-type="bibr" rid="B27">2016</xref>; Hogenkamp et al., <xref ref-type="bibr" rid="B43">2016</xref>).</p>
<p>The rewarding properties of food and increased responsivity to food cues in obesity have been widely studied (e.g., Stice et al., <xref ref-type="bibr" rid="B87">2009</xref>; Garc&#x000ED;a-Garc&#x000ED;a et al., <xref ref-type="bibr" rid="B31">2014</xref>; Pursey et al., <xref ref-type="bibr" rid="B72">2014</xref>; Alonso-Alonso et al., <xref ref-type="bibr" rid="B1">2015</xref>; Horstmann et al., <xref ref-type="bibr" rid="B45">2015a</xref>; Mathar et al., <xref ref-type="bibr" rid="B62">2016</xref>; M&#x000FC;hlberg et al., <xref ref-type="bibr" rid="B68">2016</xref>). In contrast, the differential processing of reward and punishment and reinforcement learning in a none-food context are far less understood in individuals with obesity. Coppin et al. (<xref ref-type="bibr" rid="B15">2014</xref>) presented first evidence for performance deficits in a probabilistic reinforcement learning tasks in individuals with obesity along with working memory differences between lean and obese participants. Importantly, obesity-related deficits in reinforcement learning were specific to the avoidance of negative outcomes suggesting a differential sensitivity to positive and negative feedback. Opel et al. (<xref ref-type="bibr" rid="B69">2015</xref>) reported increased neural responses in reward-related brain regions in individuals with obesity when presented with monetary gains, with no obesity-specific alterations in the processing of losses. In contrast, Balodis et al. (<xref ref-type="bibr" rid="B3">2013</xref>) observed greater functional activation in subcortical and prefrontal brain regions in individuals with obesity for the processing of both monetary gains and losses. Thus, evidence for obesity-specific deficits in reinforcement learning and differential processing of reward and punishment in obesity is still inconclusive.</p>
<p>Gender-related influences on feedback processing and learning have likewise been reported in previous studies. For example, higher performance levels in men than in women were observed in reversal learning and the well-known Iowa Gambling task (Weller et al., <xref ref-type="bibr" rid="B99">2009</xref>; Evans and Hampson, <xref ref-type="bibr" rid="B26">2015</xref>). Robinson et al. (<xref ref-type="bibr" rid="B77">2010b</xref>) report gender effects of dopamine depletion on learning from punishment with significant improvement of punishment processing after dopamine depletion in women, but not in men. In addition, in the context of feedback processing and learning gender was found to closely interact with obesity. For example, women with obesity showed a preference for risky choices despite infrequent punishment with high penalties as well as decreased behavioral adaptation after punishment in the Iowa Gambling Task (Horstmann et al., <xref ref-type="bibr" rid="B44">2011</xref>). Interestingly, this was accompanied by gender-specific correlations between markers of obesity and gray matter volume (GMV) in brain structures involved in learning, cognitive control, and goal-directed behavior.</p>
<p>The impact of weight status and gender on general heart rate variability (HRV) have long been known (e.g., Zahorska-Markiewicz et al., <xref ref-type="bibr" rid="B104">1993</xref>; Ramaekers et al., <xref ref-type="bibr" rid="B74">1998</xref>; Karason et al., <xref ref-type="bibr" rid="B49">1999</xref>; Windham et al., <xref ref-type="bibr" rid="B102">2012</xref>; Koenig and Thayer, <xref ref-type="bibr" rid="B53">2016</xref>). However, to the best of our knowledge only two studies from our own lab assessed phasic HR changes in the context of gender and obesity to date. In these studies we observed, for obese women specifically, blunted cardiac responses to social compared to monetary stimuli (Kube et al., <xref ref-type="bibr" rid="B54">2016</xref>) together with strong cardiac slowing in novel social interactions (Schrimpf et al., <xref ref-type="bibr" rid="B81">2017</xref>).</p>
<p>The third goal of our study was to explore the effects of weight status and gender on phasic HR changes in the differential processing of reward and punishment during reinforcement learning. Further we aimed at consolidating the heterogeneous previous findings on the behavioral level by a systematic assessment of obesity&#x02014;and gender-specific alterations in learning performance, behavioral adaptation and computational model parameters. We expected weight status and gender to impact on both performance and HR responses, possibly differentially for reward and punishment. However, sparsity and inconsistency of previous results, as shown above, precluded clear hypotheses for the size and direction of these effects, rendering the present assessment of these two factors more exploratory.</p>
<p>Finally, we have to consider different personal characteristics that might influence reinforcement learning from reward and punishment. Participants&#x00027; general sensitivity to reward and punishment may impact on reinforcement learning performance, given that the learning process heavily relies on the adequate evaluation of rewarding and punishing feedback. In addition, weaknesses in learning from reward and punishment and impaired adaptation of choice behavior have previously been linked to high trait impulsivity (e.g. Franken et al., <xref ref-type="bibr" rid="B30">2008</xref>), although an earlier study by the same authors did not result in conclusive evidence for this relationship (Franken and Muris, <xref ref-type="bibr" rid="B29">2005</xref>). In the same vein, it was argued that working memory capacity crucially impacts on reinforcement learning (Collins and Frank, <xref ref-type="bibr" rid="B13">2012</xref>), in particular on the processing of PE signals (Collins et al., <xref ref-type="bibr" rid="B14">2017</xref>). Therefore, participants&#x00027; working memory capacity, reward and punishment sensitivity, and trait impulsivity were taken into account in our analyses.</p>
</sec>
<sec sec-type="materials and methods" id="s2">
<title>Materials and methods</title>
<sec>
<title>Participants</title>
<p>Sixty Caucasian participants, aged between 18 and 36 years, were initially invited to our experiment. All participants were right-handed, had normal or corrected-to-normal vision, and were grouped according to BMI into a group of participants with (BMI &#x02265;30 kg/m<sup>2</sup>, &#x0003C;45 kg/m<sup>2</sup>) and without (BMI &#x02265;18.5 kg/m<sup>2</sup>, &#x0003C;25 kg/m<sup>2</sup>) obesity. Participants were recruited from the participant database of the Max Planck Institute for Human Cognitive and Brain Sciences, Leipzig, Germany. All participants provided written informed consent prior to participation. The study complies with the ethical standards of the Declaration of Helsinki and was approved by the ethics committee of the University of Leipzig.</p>
<p>All participants underwent an initial telephone screening to evaluate inclusion and exclusion criteria. Exclusion criteria were a history of neurological or neuropsychiatric disorders, current smoking, recent or current dieting, use of drugs, psychoactive medication, or medication influencing the autonomic nervous system. These exclusion criteria were chosen to avoid confounding alterations in reinforcement processing due neuropsychiatric symptomatology and medication (Etkin and Wager, <xref ref-type="bibr" rid="B25">2007</xref>; Wittmann and D&#x00027;Esposito, <xref ref-type="bibr" rid="B103">2015</xref>), smoking status (Martin et al., <xref ref-type="bibr" rid="B61">2014</xref>), and hunger (Symmonds et al., <xref ref-type="bibr" rid="B90">2010</xref>; Levy et al., <xref ref-type="bibr" rid="B58">2013</xref>). Participants reporting hyper- or hypothyroidism were excluded, since these conditions may affect their baseline cardiac responses as well as body weight status (Bratusch-Marrain et al., <xref ref-type="bibr" rid="B8">1978</xref>; Cacciatori et al., <xref ref-type="bibr" rid="B11">2000</xref>; Tzotzas et al., <xref ref-type="bibr" rid="B92">2000</xref>). As previous studies have shown that hypertension may be associated with altered baseline cardiac responses (Schroeder et al., <xref ref-type="bibr" rid="B82">2003</xref>; Kim et al., <xref ref-type="bibr" rid="B50">2016</xref>), we excluded participants who reported hypertension during the telephone screening or exhibited values exceeding the range for normal or high normal blood pressure (Mancia et al., <xref ref-type="bibr" rid="B60">2013</xref>) in a manual examination after the experiment. Further, a depressive symptomatology has been found to be associated with altered HR responses to the presentation of rewarding stimuli (Brinkmann and Franzen, <xref ref-type="bibr" rid="B9">2013</xref>, <xref ref-type="bibr" rid="B10">2017</xref>). Therefore, we measured the current depressive symptomatology using Beck&#x00027;s Depression Inventory-Short Form (BDI-SF, Beck and Steer, <xref ref-type="bibr" rid="B4">1993</xref>) and excluded participants with a BDI-SF &#x0003E; 10. Finally, as even moderate physical exercise impacts on measures of HR and HRV (Rennie et al., <xref ref-type="bibr" rid="B75">2003</xref>; Hottenrott et al., <xref ref-type="bibr" rid="B47">2006</xref>), we excluded participants with more than 3 h per week of regular cardiovascular training.</p>
<p>Upon participation, a total of 12 participants had to be excluded due to an excessive number of miss trials during the experiment (5), insufficient task comprehension identified during a debriefing interview (6) and technical problems (1). Thus, our final sample consisted of 48 participants (mean age: 25.9 &#x000B1; 4.37 years; range between 20 and 36 years) including 24 participants with obesity (BMI &#x0003D; 35.59 kg/m<sup>2</sup> &#x000B1; 3.39 kg/m<sup>2</sup>, range 30.68&#x02013;43.33 kg/m<sup>2</sup>, 12 female) and 24 lean participants (BMI &#x0003D; 22.18 kg/m<sup>2</sup> &#x000B1; 1.37 kg/m<sup>2</sup>, range 19.83&#x02013;24.09 kg/m<sup>2</sup>, 12 female). Groups were matched for age and level of educational background. For the latter we chose years of scholastic education as a comparable objective variable. All but two participants finished at least 12 years of scholastic education, which in the German educational system is the prerequisite to enter university to receive higher education. Two participants finished secondary school after 10 years followed by vocational training, which represents the second highest level of scholastic education.</p>
</sec>
<sec>
<title>Experimental task</title>
<p>Participants performed a probabilistic reinforcement learning task adapted from Kim et al. (<xref ref-type="bibr" rid="B51">2006</xref>) and B&#x000F3;di et al. (<xref ref-type="bibr" rid="B6">2009</xref>). The task consisted of 240 trials. In each trial participants were presented with a pair of symbols and had to choose one of them by button press. Three different pairs of symbols were included in the experiment: (1) one pair signaled the possibility of winning 50 points or receiving no outcome (80 reward/gain trials), (2) one pair signaled the possibility of losing 50 points or receiving no outcome (80 punishment/loss trials), and (3) one pair was associated with a neutral outcome signaling neither gain nor loss (80 neutral trials), see Figure <xref ref-type="fig" rid="F1">1</xref>. In each pair, one symbol was associated with a higher probability of receiving the respective outcome: In gain trials, the advantageous symbol was associated with a 70% probability of winning 50 points and lead to no outcome in only 30% of the trials in which it was chosen. The disadvantageous symbol was associated with only a 30% probability of winning 50 points and led to no outcome in 70% of the trials in which it was chosen. Similarly, in loss trials the advantageous symbol was associated with a 70% probability of avoiding to lose 50 points, while the other symbol had a loss avoidance probability of only 30%. In the neutral control condition, the two symbols likewise had a 70 and 30% probability of seeing neutral feedback, and 30 and 70% probability of no outcome, respectively. Symbols were randomly assigned to a given trial type, and trial order was randomized in blocks of 30 trials to ensure a roughly equal number of trials per condition in each stage of the experiment.</p>
<fig id="F1" position="float">
<label>Figure 1</label>
<caption><p>Experimental task. Example trial and task structure with reward/gain and punishment/loss probabilities of the reinforcement learning task.</p></caption>
<graphic xlink:href="fnins-11-00598-g0001.tif"/>
</fig>
<p>Note that with this task design, participants could maximize their overall task performance by choosing the advantageous symbol in both the reward and the punishment condition, i.e., by learning to select the high probability reward symbol in reward trials and the high probability punishment avoidance symbol in punishment trials. Choices during neutral trials did not affect task performance. Importantly, independence of positive and negative feedback in this task design additionally enabled us to determine individual differences in learning from the two feedback valences.</p>
<p>Trial timing in an example trial for the reward condition is displayed in Figure <xref ref-type="fig" rid="F1">1</xref>. The pair of stimulus symbols was presented for a maximum of 1,500 ms and participants were asked to select one of them. Once the symbol was selected, the chosen option was highlighted for 1,000 ms and a blank screen followed for a 1,000 ms delay period. Thereafter, the outcome was presented for 2,000 ms. If the participants received no outcome, a fixation cross was shown instead. If the participants did not press the button or were too slow, the trial was aborted and the text &#x0201C;zu langsam!&#x0201D; (too slow) appeared on the screen. These trials were dismissed from further analyses (1.5% of all trials). Each trial was followed by an inter-trial-interval of 1,600&#x02013;2,200 ms.</p>
<p>Prior to the experiment, participants were instructed about the task and performed a practice run of 12 trials, four trials in each condition. The instructions included the information that two symbols would be presented in each trial and the task was to select one of them. Depending on their choice participants would win 50 points, lose 50 points, receive a financially neutral outcome or no feedback. Participants were informed that the task comprised of three trial conditions and that in each trial condition one symbol had a higher probability of leading to an advantageous outcome. However, they did not know which symbol was associated with a particular outcome. In addition, participants were informed that their net gain would be transformed into a monetary bonus at the end of the experiment. Upon completion of all tasks and questionnaires participants were debriefed about the aim of the study.</p>
</sec>
<sec>
<title>Experimental procedure</title>
<p>Experiments were performed in a sound proof room that was artificially lid with the blinds closed. Upon arrival, participants were explained the procedure and comfortably seated on a chair in front of a computer screen. They first performed the working memory test. Participants were then prepared for electrocardiogram (ECG) recording. To allow ECG to stabilize after preparation, participants filled in the first questionnaire. This was followed by a 5 min baseline recording of ECG data and the experimental task, including pre- and post-ratings of the symbols. After the experimental task was finished, participants filled out a second set of questionnaires, and were debriefed about the experiment. Finally, participants&#x00027; current height, weight, and blood pressure were measured. The entire experiment lasted for &#x0007E;2 h. Participants received reimbursement of 7 Euro per hour and additional bonus of 2.86 Euro on average, according to the score reached by the end of the experiment.</p>
</sec>
<sec>
<title>Data acquisition</title>
<sec>
<title>ECG data</title>
<p>The ECG was continuously recorded during the task with a sampling rate of 500 Hz using BioPac 3.7.7 and the MP35 recording unit. In order to ensure that participants show typical and healthy cardiac responses at rest, ECG was also recorded for HRV analysis during a 5 min resting period before the start of the experiment. Three Ag/AgCl ECG electrodes (Nessler Swaromed) were placed below the right collar bone about 10 cm from the sternum, on the left side between the lower two ribs, and on the right lower abdomen. ECG data analysis was carried out using customized Matlab-based scripts (Matlab R2013b, The MathWorks, Sherborn, MA, USA) for R-peak detection and artifact correction. Automatic R-peak detection identified all stationary data points that exceeded 20% of the global ECG maximum and were preceded by data points with a first derivative 1.5 times larger than the global ECG maximum. A median template of the QRS-complex around the detected R-peaks was calculated, and QRS complexes with a cross correlation coefficient larger than 0.8 were selected. All automatically detected R-peaks were visually inspected to ensure correct R-peak detection and manually corrected where necessary. Inter-beat-intervals (IBI) were calculated as time difference between two subsequent R-peaks. IBIs deviating more than 3.5 SDs from the session&#x00027;s mean IBI or more than 50% from the preceding IBI were identified as artifacts and replaced by the session&#x00027;s mean IBI length. Note that this approach differs from the often applied interpolation by neighboring IBIs. However, we avoided any interpolation from neighboring IBIs in ECG data modeling, as the statistical analysis of phasic IBI changes crucially depends on the direct comparison of neighboring IBIs. Interpolating corrupted IBIs by their neighbors might therefore compromise the validity of the statistical analysis. Across participants only 0.37% of all IBIs (421 out of 112,016) were identified as artifacts with an average of 0.36% of IBIs per person. The largest number of IBIs replaced for an individual participant amounted to 66 out of 3,034. These very few artifacts were unlikely to significantly impact on subsequent statistical analyses.</p>
</sec>
<sec>
<title>Personality traits, working memory scores, and ratings</title>
<p>Four potential influencing factors were regarded in the behavioral analysis and assessed for each participant prior to or after the task: participants&#x00027; responsiveness to reward, responsiveness to punishment, impulsivity, and working memory capacity. The first three factors were assessed by means of two questionnaires, the BIS/BAS (Carver and White, <xref ref-type="bibr" rid="B12">1994</xref>) and the UPPS Impulsive Behavior Scale (Whiteside and Lynam, <xref ref-type="bibr" rid="B101">2001</xref>). The BIS/BAS captures two general motivational systems that underlie behavior. The Behavioral Inhibition System (BIS) represents an aversive motivational system that is sensitive to punishment and reward omission. The Behavioral Activation System (BAS) reflects an appetitive motivational system which is sensitive to reward and the avoidance of punishment. Note that the internal consistency of the three BAS factors drive, fun seeking, and reward responsiveness is still under debate for the German version of the questionnaire that was used in our experiment (Strobel et al., <xref ref-type="bibr" rid="B88">2001</xref>; Mueller et al., <xref ref-type="bibr" rid="B67">2013</xref>). Observed effects regarding these factors should thus be treated with caution.</p>
<p>The UPPS is designed to assess distinct personality facets associated with impulsive behavior: urgency, (lack of) premeditation, (lack of) perseverance, and sensation seeking. These four subscales possess very good internal consistency in the German version of the UPPS (Schmidt et al., <xref ref-type="bibr" rid="B79">2008</xref>).</p>
<p>Possible inter-individual performance differences due to visual working memory capacity, were assessed in the German version of the revised Wechsler Memory Scale (WMS-R), subtest Figural Memory (Wechsler, <xref ref-type="bibr" rid="B98">1987</xref>; H&#x000E4;rting et al., <xref ref-type="bibr" rid="B42">2000</xref>).</p>
<p>Immediately before and after the learning task, we obtained subjective valence and arousal ratings for each symbol to determine changes in affective responses toward the stimuli. Here, each cue was presented individually and rated according to valence and arousal on 9-point Self-Assessment Manikin visual analog scales (Bradley and Lang, <xref ref-type="bibr" rid="B7">1994</xref>). This enabled us to investigate task-induced differential changes in the evaluation of advantageous and disadvantageous symbols.</p>
</sec>
</sec>
<sec>
<title>Data analysis</title>
<sec>
<title>Computational model of learning behavior</title>
<p>Trial-wise PEs, subject- and condition-specific learning rates and choice consistency estimates were derived from a computational reinforcement model. The model is an implementation of the Q-learning algorithm (Watkins and Dayan, <xref ref-type="bibr" rid="B97">1992</xref>). It was previously applied in a comparable implicit learning paradigm in healthy individuals and clinical populations, where it was shown to adequately capture reinforcement learning tasks based on time-invariant probabilistic stimulus-outcome associations (Mathar et al., <xref ref-type="bibr" rid="B64">2017b</xref>). In more detail, the model consists of six input nodes <italic>I</italic><sub><italic>i</italic>&#x0003D;1,&#x02026;,6</sub> with weighted connections to two output nodes (<italic>Q</italic>-values) <italic>Q</italic><sub><italic>j</italic>&#x0003D;1,2</sub> that represent the presence or absence of the six possible symbols <italic>i</italic> (three pairs of symbols) and the two possible outcomes <italic>j</italic> in each condition, respectively. On each trial, activity of the output nodes is computed as</p>
<disp-formula id="E1"><label>(1)</label><mml:math id="M1"><mml:mtable class="eqnarray" columnalign="right center left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>Q</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mtext>&#x000A0;</mml:mtext><mml:msub><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mtext>&#x000A0;</mml:mtext><mml:msub><mml:mrow><mml:mi>I</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <italic>w</italic><sub><italic>ij</italic></sub> represents the weight connecting input node <italic>I</italic><sub><italic>i</italic></sub> and output node <italic>Q</italic><sub><italic>j</italic></sub>. Weights are initialized to 0.25, representing equal distribution of initial weights between the four connections that can be updated within one trial (connections from two stimulus symbols at a time to the two outcomes). Weights are updated in each trial <italic>k</italic> by means of</p>
<disp-formula id="E2"><label>(2)</label><mml:math id="M2"><mml:mtable class="eqnarray" columnalign="right center left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>k</mml:mi><mml:mo>&#x0002B;</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>k</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x0002B;</mml:mo><mml:msup><mml:mrow><mml:mi>&#x003B1;</mml:mi></mml:mrow><mml:mrow><mml:mi>r</mml:mi><mml:mo>/</mml:mo><mml:mi>p</mml:mi><mml:mo>/</mml:mo><mml:mi>n</mml:mi></mml:mrow></mml:msup><mml:msub><mml:mrow><mml:mi>S</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>-</mml:mo><mml:msub><mml:mrow><mml:mi>Q</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:msub><mml:mrow><mml:mi>I</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where <italic>R</italic><sub><italic>j</italic></sub> encodes the actual outcome in this trial and <italic>S</italic><sub><italic>j</italic></sub> represents the participant&#x00027;s choice. The latter is included for allowing the model to simulate the behavior of the individual participant rather than optimal learning.</p>
<p>To differentially assess learning from reward (potential gains) and punishment (potential losses), we fitted three independent learning rates for the reward &#x003B1;<sup><italic>r</italic></sup>, punishment &#x003B1;<sup><italic>p</italic></sup>, and neutral condition &#x003B1;<sup><italic>n</italic></sup>, respectively. In reinforcement learning, a learning rate reflects how strongly new experiences in one trial impact on the participant&#x00027;s knowledge acquired over all previous trials. For each participant, the three individual learning rates were determined that minimized the sum of squared differences between the model&#x00027;s output and the participant&#x00027;s choice:</p>
<disp-formula id="E3"><label>(3)</label><mml:math id="M3"><mml:mtable class="eqnarray" columnalign="right center left"><mml:mtr><mml:mtd><mml:msub><mml:mrow><mml:mo>&#x02211;</mml:mo></mml:mrow><mml:mrow><mml:mi>j</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msub><mml:msup><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>S</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msub><mml:mo>-</mml:mo><mml:msub><mml:mrow><mml:mi>Q</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup><mml:mo>&#x02192;</mml:mo><mml:mi>m</mml:mi><mml:mi>i</mml:mi><mml:mi>n</mml:mi><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>with <italic>j</italic> &#x0003D; 1, 2 and <italic>k</italic> again marking the trial number. In a subsequent step, we modeled the probability for each participant&#x00027;s choices of a particular symbol to follow a softmax distribution:</p>
<disp-formula id="E4"><label>(4)</label><mml:math id="M4"><mml:mtable class="eqnarray" columnalign="right center left"><mml:mtr><mml:mtd><mml:mi>P</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>c</mml:mi><mml:mi>h</mml:mi><mml:mi>o</mml:mi><mml:mi>i</mml:mi><mml:mi>c</mml:mi><mml:mi>e</mml:mi><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>S</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>|</mml:mo><mml:msub><mml:mrow><mml:mi>Q</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:msub><mml:mrow><mml:mi>Q</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>e</mml:mi><mml:mi>x</mml:mi><mml:mi>p</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>&#x003B2;</mml:mi><mml:msub><mml:mrow><mml:mi>Q</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>e</mml:mi><mml:mi>x</mml:mi><mml:mi>p</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>&#x003B2;</mml:mi><mml:msub><mml:mrow><mml:mi>Q</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>&#x0002B;</mml:mo><mml:mi>e</mml:mi><mml:mi>x</mml:mi><mml:mi>p</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>&#x003B2;</mml:mi><mml:msub><mml:mrow><mml:mi>Q</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow></mml:mfrac><mml:mi>w</mml:mi><mml:mi>i</mml:mi><mml:mi>t</mml:mi><mml:mi>h</mml:mi><mml:mtext>&#x000A0;</mml:mtext><mml:mi>j</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mtext>&#x000A0;</mml:mtext><mml:mn>2</mml:mn><mml:mo>,</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>where the parameter &#x003B2; reflects the consistency of choices made by the participant. That is, the parameter reflects how deterministic or stochastic the participant behaves over the course of the experiment, with high &#x003B2;-values representing more stochastic or inconsistent behavior.</p>
<p>Model fitting and estimation of all parameters was accomplished by non-linear optimization. Recall that the PE in each trial encodes the discrepancy between expected and actual outcome. Thus, after model fitting the prediction error <italic>PE</italic><sub><italic>k</italic></sub> for trial <italic>k</italic> can be directly derived from Equation (2) as</p>
<disp-formula id="E6"><label>(5)</label><mml:math id="M6"><mml:mtable class="eqnarray" columnalign="right center left"><mml:mtr><mml:mtd><mml:mi>P</mml:mi><mml:msub><mml:mrow><mml:mi>E</mml:mi></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>S</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msub><mml:mo>-</mml:mo><mml:msub><mml:mrow><mml:mi>Q</mml:mi></mml:mrow><mml:mrow><mml:mi>j</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:msub><mml:mrow><mml:mi>I</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>k</mml:mi></mml:mrow></mml:msub><mml:mo>.</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>Prior to statistical analysis of model parameters, model adequacy was assessed in two ways. First, the model&#x00027;s choice consistency parameter &#x003B2; was regressed against the overall number of switches between choices exhibited by the participant. A strong regression signifies that, across subjects, the model adequately captured participants&#x00027; behavior, because if the model correctly reproduces participants&#x00027; actual behavior, then a model&#x00027;s choice consistency (small &#x003B2;) should go along with few switches made by a participant, while inconsistent choice behavior of the model (i.e., large &#x003B2;) should entail large number of switches by the participant. Second, model fit was compared across participant groups by means of the Bayesian Information Criterion (BIC, Schwarz, <xref ref-type="bibr" rid="B85">1978</xref>), as comparable model fit is a prerequisite for parameter comparability.</p>
</sec>
<sec>
<title>Autonomic responses</title>
<p>In our event-related HR analysis we closely followed the procedures applied in previous assessments of phasic cardiac concomitants of stimulus and feedback processing (e.g., Somsen et al., <xref ref-type="bibr" rid="B86">2000</xref>; Crone et al., <xref ref-type="bibr" rid="B19">2003</xref>; Van Der Veen et al., <xref ref-type="bibr" rid="B93">2004</xref>; Groen et al., <xref ref-type="bibr" rid="B39">2007</xref>). For the event-related analysis of stimulus processing, four IBIs were extracted around stimulus presentation: IBI 0 was measured at the time of stimulus presentation and was preceded by IBI &#x02212;1 and immediately followed by IBIs 1 and 2. All stimulus-related IBIs were referenced to a statistically independent IBI &#x02212;2 prior to trial start. Statistical analyses of this reference IBI revealed no significant valence, gender, or obesity effect (repeated measures ANOVAs with within-subject factor valence (reward, punishment, neutral) and between-subject factors gender and obesity; all <italic>p</italic> &#x0003E; 0.282).</p>
<p>For the event-related analysis of feedback processing, five IBIs were extracted: IBI 0 was measured at the time of feedback presentation and was preceded by IBI &#x02212;1 and immediately followed by IBIs 1, 2, and 3. In order to marginalize the impact of differential stimulus processing on the feedback-related analysis, all IBIs were now referenced to the IBI &#x02212;2 prior to feedback presentation. This ensures independence of IBI changes at feedback presentation from IBI changes at stimulus presentation, as a reference IBI after stimulus presentation effectively functions as a new &#x0201C;baseline&#x0201D; preceding feedback presentation. Note that for the purpose of plotting responses to stimuli and feedback on a common scale in <bold>Figure 3</bold>, in this plot all IBIs from stimulus presentation to HR recovery after feedback presentation are referenced to a common IBI-2 prior to stimulus presentation and named IBI 0 (presentation of stimulus) to IBI 7.</p>
<p>In order to assess learning-induced effects on performance over the course of the experiment, experimental trials were divided into four task blocks of 60 trials each. Learning-induced effects on phasic IBI changes were expected to emerge later than behavioral adaptation. They were thus assessed by comparison of the first and the second experimental half, containing trials 1 to 120 and trials 121&#x02013;240, respectively.</p>
<p>Finally, a potential learning-induced shift in heart beat responsiveness from the presentation of feedback to the presentation of stimuli was directly investigated based on the mean area under the curve (AUC) that describes changes in IBI length following stimulus and feedback presentation. Specifically, for each subject, a trapezoid was calculated, representing the AUC of changes in IBI length from IBI 0 to IBI 2 after the presentation of a stimulus and the presentation of feedback, respectively. Note that reference IBIs for stimulus-related and feedback-related IBIs were identical to the independent analyses of stimulus and feedback processing in order to ensure comparability of AUCs across the two event types. Mean AUCs were then submitted to a repeated measures ANOVA containing event type (stimulus, feedback), experimental half (1st half, 2nd half), and valence (reward, punishment) as within-subject and gender and obesity as between-subject factors.</p>
</sec>
<sec>
<title>Relationship of the PE signal and autonomic responses</title>
<p>Subsequent to modeling each participant&#x00027;s learning behavior, we analyzed the relationship between the obtained trial-wise PEs and the relative IBI length following feedback presentation. For this, we employed a general linear model (GLM) and subsequent statistical evaluation of the GLM parameters. Specifically, for each subject, the vector of PEs after positive and after negative feedback was modeled as</p>
<disp-formula id="E7"><mml:math id="M7"><mml:mtable columnalign="left"><mml:mtr><mml:mtd><mml:mi>P</mml:mi><mml:msub><mml:mrow><mml:mi>E</mml:mi></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>b</mml:mi></mml:mrow><mml:mrow><mml:mi>c</mml:mi></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:mtext>&#x000A0;</mml:mtext><mml:msub><mml:mrow><mml:mi>b</mml:mi></mml:mrow><mml:mrow><mml:mn>0</mml:mn></mml:mrow></mml:msub><mml:mi>I</mml:mi><mml:mi>B</mml:mi><mml:mi>I</mml:mi><mml:msub><mml:mrow><mml:mn>0</mml:mn></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:mtext>&#x000A0;</mml:mtext><mml:msub><mml:mrow><mml:mi>b</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mi>I</mml:mi><mml:mi>B</mml:mi><mml:mi>I</mml:mi><mml:msub><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:mtext>&#x000A0;</mml:mtext><mml:msub><mml:mrow><mml:mi>b</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mi>I</mml:mi><mml:mi>B</mml:mi><mml:mi>I</mml:mi><mml:msub><mml:mrow><mml:mn>2</mml:mn></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msub><mml:mo>&#x0002B;</mml:mo><mml:mtext>&#x000A0;</mml:mtext><mml:msub><mml:mrow><mml:mi>&#x003B5;</mml:mi></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msub></mml:mtd></mml:mtr></mml:mtable></mml:math></disp-formula>
<p>for all trials <italic>k</italic> that led to positive or negative feedback respectively, with the PE derived from the computational learning model as dependent variable, IBI 0, IBI 1, and IBI 2 as independent variables, the corresponding vector of coefficients <italic>b</italic> and the error vector &#x003B5;. Across subjects, coefficient estimates corresponding to IBI 0, IBI 1 and IBI 2 were subjected to a one sample <italic>T</italic>-test. Coefficients with a significant difference from zero mark a predictive effect of the corresponding independent variable on the size of the PE. In other words, across participants, PE and IBI are systematically associated for any IBI with a GLM coefficient that significantly differs from zero. In order to compare predictive effects across the three IBIs as well as across conditions, GLM coefficients were subsequently subjected to a repeated measures ANOVA with IBI (IBI 0, IBI 1, IBI 2) and valence (reward, punishment) as within-subject factors. For the comparison across IBIs, standardized coefficients were used.</p>
</sec>
</sec>
<sec>
<title>Statistical methods</title>
<p>All acquired and modeled data as well as their hypothesized interdependencies were statistically analyzed using IBM SPSS Statistics 22.0 (IBM Corp., Armonk, NY, USA). For all statistical tests we assume statistical significance for <italic>p</italic> &#x0003C; 0.05. For each analysis, statistical tests were chosen depending on the nature and distribution of the data as follows. Group differences (lean vs. obese, female vs. male) for normally distributed data in demographics, questionnaire scores, performance measures, BIC-values, and model parameters were analyzed by univariate ANOVAs with obesity and gender as fixed between-subject factors. For normally distributed data we report mean and standard deviation. Mann-Whitney-U-Tests were applied when the assumption of normality was violated as assessed by Shapiro-Wilk test. Here, we report medians and [min, max] of the data or, in cases where the full range of possible values was covered by the results, [25th, 75th percentiles]. Pairwise <italic>post-hoc</italic> comparisons were calculated to assess origin and directionality of interaction effects observed in univariate or repeated measures analyses of variance.</p>
<p>Across participants, performance differences were compared between experimental conditions and between task blocks by related samples Friedman&#x00027;s Two-Way Analysis of Variance by Ranks for three or more conditions or task blocks, and by Wilcoxon signed rank tests for two conditions, respectively. Differences between conditions in the number of switches were statistically assessed by a sign test, as the assumption of the Wilcoxon signed rank test for a symmetrically shaped distribution of differences was not met. Reaction times were analyzed by repeated measures ANCOVA with between-subject factors obesity and gender, within-subject factor valence (reward, punishment, neutral) and task block (blocks 1&#x02013;4). Age was included as covariate of no interest.</p>
<p>Differences in valence and arousal ratings between symbols prior to the task were assessed by repeated measures ANOVAs with symbol as within-subject factor and obesity and gender as between-subject factors. Task induced changes in valence and arousal ratings were analyzed by repeated measures ANOVAs with time point (pre/post task) as within-subject factor and obesity and gender as between-subject factors. Bivariate correlations between questionnaire scores and working memory capacity with performance measures were determined by Pearson&#x00027;s correlation coefficients. Normality of the data was ensured by Shapiro-Wilk test.</p>
<p>Differences in phasic IBI were statistically evaluated by repeated measures ANOVAs. Specifically, mean IBI differences in response to stimulus presentation were statistically evaluated using a repeated measures ANOVA with valence (reward, punishment), experimental half (1st half, 2nd half), and IBI (four levels; IBI &#x02212;1, IBI 0, IBI 1, IBI 2) as within-subjects factors and obesity and gender as between-subjects factors. Note that the &#x0201C;factor experimental half&#x0201D; was included in the ANOVAs to identify autonomic reactions that might only be present at the beginning or toward the end of learning. The specific analysis of a potential learning-induced shift in autonomic responsiveness is described below. For the analysis of IBI differences in response to feedback, the within-subject factor IBI consisted of five levels: IBI &#x02212;1, IBI 0, IBI 1, IBI 2, IBI 3. To identify the underlying cause in IBI differences, e.g., differences in deceleration or recovery speed between conditions or participant groups, changes between neighboring IBIs were assessed. In other words, differences between IBI &#x02212;1 and IBI 0, between IBI 0 and IBI 1 and so forth were calculated and statistically compared across any interacting effect e.g., between genders or positive and negative feedback trials. Graphically, this is reflected in the steepness of the slope between two neighboring IBIs.</p>
<p>All pairwise <italic>post-hoc</italic> tests and all statistical tests involving dependent data were Bonferroni corrected for multiple comparisons. The latter included, for example, all tests involving the number of switches, all correlations of the same performance measure with personality traits etc. In cases where correction for multiple comparisons was required, we report only those <italic>p</italic>-values as significant that are below the adjusted significance threshold and report the applied number of tests as correction factor (CF), e.g., <italic>p</italic>-values below the adjusted threshold of <italic>p</italic> &#x0003C; 0.025 and the correction factor CF 2 in case of two tests on dependent data.</p>
</sec>
</sec>
<sec sec-type="results" id="s3">
<title>Results</title>
<sec>
<title>Demographics</title>
<p>Descriptive statistics of participants&#x00027; demographic characteristics are reported in Table <xref ref-type="table" rid="T1">1</xref>. As intended, across groups participants did not differ with respect to age and educational background. Lean and obese participants significantly differed in weight, BMI, and waist-to-hip ratio. Male and female subjects significantly differed in height, weight, and waist-to-hip ratio but, importantly, not in BMI distribution. Gender and weight status can thus be regarded as independent factors in the statistical analysis. In the same vein, baseline HR did not significantly differ between groups and HRV analysis at rest revealed typical patterns of cardiac activity (<xref ref-type="supplementary-material" rid="SM1">Supplementary Material</xref>), ruling out any impact of those factors on the subsequently observed effects in phasic cardiac responses.</p>
<table-wrap position="float" id="T1">
<label>Table 1</label>
<caption><p>Descriptive statistics.</p></caption>
<table frame="hsides" rules="groups">
<thead><tr>
<th/>
<th valign="top" align="center" colspan="2" style="border-bottom: thin solid #000000;"><bold>LEAN</bold></th>
<th valign="top" align="center" colspan="2" style="border-bottom: thin solid #000000;"><bold>OBESE</bold></th>
<th valign="top" align="center" colspan="2" style="border-bottom: thin solid #000000;"><italic><bold>p</bold></italic></th>
</tr>
<tr>
<th/>
<th valign="top" align="center"><bold>Male</bold></th>
<th valign="top" align="center"><bold>Female</bold></th>
<th valign="top" align="center"><bold>Male</bold></th>
<th valign="top" align="center"><bold>Female</bold></th>
<th valign="top" align="center"><bold>Factor obesity</bold></th>
<th valign="top" align="center"><bold>Factor gender</bold></th>
</tr>
</thead>
<tbody>
<tr>
<td valign="top" align="left">Age (years)</td>
<td valign="top" align="center">26.2 (5.78)</td>
<td valign="top" align="center">25.0 (4.41)</td>
<td valign="top" align="center">26.7 (3.2)</td>
<td valign="top" align="center">26.0 (4.11)</td>
<td valign="top" align="center">0.564</td>
<td valign="top" align="center">0.482</td>
</tr>
<tr>
<td valign="top" align="left">Years of education</td>
<td valign="top" align="center">13 (13-13)</td>
<td valign="top" align="center">13 (13-13)</td>
<td valign="top" align="center">13 (10-13)</td>
<td valign="top" align="center">13 (10-13)</td>
<td valign="top" align="center">0.187</td>
<td valign="top" align="center">0.657</td>
</tr>
<tr>
<td valign="top" align="left">Height (m)</td>
<td valign="top" align="center">1.80 (0.04)</td>
<td valign="top" align="center">1.71 (0.06)</td>
<td valign="top" align="center">1.80 (0.7)</td>
<td valign="top" align="center">1.67 (0.06)</td>
<td valign="top" align="center">0.206</td>
<td valign="top" align="center"><bold>&#x0003C;0.001</bold></td>
</tr>
<tr>
<td valign="top" align="left">Weight (kg)</td>
<td valign="top" align="center">73.37 (4.77)</td>
<td valign="top" align="center">63.75 (6.57)</td>
<td valign="top" align="center">115.24 (15.54)</td>
<td valign="top" align="center">99.37 (9.83)</td>
<td valign="top" align="center"><bold>&#x0003C;0.001</bold></td>
<td valign="top" align="center"><bold>&#x0003C;0.001</bold></td>
</tr>
<tr>
<td valign="top" align="left">BMI</td>
<td valign="top" align="center">22.63 (1.20)</td>
<td valign="top" align="center">21.73 (1.44)</td>
<td valign="top" align="center">35.59 (3.24)</td>
<td valign="top" align="center">35.61 (3.68)</td>
<td valign="top" align="center"><bold>&#x0003C;0.001</bold></td>
<td valign="top" align="center">0.565</td>
</tr>
<tr>
<td valign="top" align="left">WHR (cm)</td>
<td valign="top" align="center">0.82 (0.04)</td>
<td valign="top" align="center">0.75 (0.04)</td>
<td valign="top" align="center">0.95 (0.05)</td>
<td valign="top" align="center">0.84 (0.05)</td>
<td valign="top" align="center"><bold>&#x0003C;0.001</bold></td>
<td valign="top" align="center"><bold>&#x0003C;0.001</bold></td>
</tr>
<tr>
<td valign="top" align="left">HR (beats per min)</td>
<td valign="top" align="center">66.50 (8.74)</td>
<td valign="top" align="center">65.33 (7.5)</td>
<td valign="top" align="center">65.17 (10.53)</td>
<td valign="top" align="center">67.17 (9.78)</td>
<td valign="top" align="center">0.925</td>
<td valign="top" align="center">0.876</td>
</tr>
</tbody>
</table>
<table-wrap-foot>
<p><italic>Distribution of gender, age, level of education, height, weight, body mass index (BMI), waist-to-hip ratio (WHR), and baseline heart rate in male and female participants with and without obesity. Values represent mean and standard deviation except for years of education [median (min-max)]. Group differences were determined by univariate ANOVA with obesity and gender as fixed between-subject factors. Significant group effects at p &#x0003C; 0.05 are marked in bold</italic>.</p>
</table-wrap-foot>
</table-wrap>
</sec>
<sec>
<title>Behavioral analysis</title>
<p>Addressing hypotheses 1 and 2, we first analyzed participants&#x00027; task performance in the reinforcement learning task according to the overall score achieved and according to the number of advantageous choices made over the course of the experiment. In addition, we defined a learning criterion for the reward and punishment condition that allowed us to assess speed of learning as follows: A participant had successfully learned the task, if he or she chose the symbol with high probability of receiving a reward and with high probability of avoiding a punishment in 9 out of 10 consecutive trials in the reward and punishment condition, respectively.</p>
<p>Supporting hypothesis 1, all participants increased their scores from the initial 2,000 points with final scores ranging from 2,150 to 3,550 points. The number of advantageous choices significantly increased over the four experimental blocks for the reward and the punishment condition [choose reward: <inline-formula><mml:math id="M8"><mml:msubsup><mml:mrow><mml:mi>&#x003C7;</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>3</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msubsup></mml:math></inline-formula> &#x0003D; 53.975, <italic>p</italic> &#x0003C; 0.0005; avoid punishment <inline-formula><mml:math id="M9"><mml:msubsup><mml:mrow><mml:mi>&#x003C7;</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mn>3</mml:mn></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msubsup></mml:math></inline-formula> &#x0003D; 50.940, <italic>p</italic> &#x0003C; 0.0005, Figure <xref ref-type="fig" rid="F2">2A</xref>]. No difference across blocks was observed in the neutral condition (<italic>p</italic> &#x0003D; 0.49). Both number of advantageous reward and punishment choices were significantly higher than the number of neutral choices with high probability feedback (reward: z &#x0003D; 5.32, <italic>p</italic> &#x0003C; 0.0001; punishment z &#x0003D; 5.03, <italic>p</italic> &#x0003C; 0.0001), pointing at successful learning from both reward and punishment. However, in line with our hypothesis 2, the number of advantageous choices was significantly higher in reward compared to punishment trials (z &#x0003D; 2.470, <italic>p</italic> &#x0003D; 0.014), reflecting an increased influence of positive compared to negative reinforcement during learning.</p>
<fig id="F2" position="float">
<label>Figure 2</label>
<caption><p>Behavioral results. Top: Number of advantageous choices <bold>(A)</bold> and reaction times <bold>(B)</bold> for the reward, punishment and neutral condition; bottom: Significant gender effects in the number of advantageous choices <bold>(C)</bold> and in the overall number of switch <bold>(D)</bold> trials for the reward and punishment condition. Advantageous choices refer to trials where participants chose the symbol with the higher probability for gaining a reward or avoiding a punishment in the reward and punishment condition, respectively. Switch trials refers to those trials where participants changed their choices from one symbol in the previous trial of this condition to the other symbol in the current trial. Statistically significant differences at <italic>p</italic> &#x0003C; 0.05 are marked with <sup>&#x0002A;</sup>.</p></caption>
<graphic xlink:href="fnins-11-00598-g0002.tif"/>
</fig>
<p>Adding to the differences in task performance across conditions, we observed a statistically significant difference in the time point of reaching the learning criterion: Participants reached the learning criterion on average after 14 [25th and 75th percentile: (10,26)] reward trials, but only after 23 [25th and 75th percentile: (15,35)] punishment trials (z &#x0003D; &#x02212;1.98, <italic>p</italic> &#x0003D; 0.047). Note that four participants did not reach the learning criterion in one or both conditions. These four participants were excluded from all analysis involving the learning criterion.</p>
<p>Performance scores and number of advantageous choices significantly correlated in a negative way with the overall number of switches in both reward (score <italic>r</italic> &#x0003D; &#x02212;0.787, advantageous choices: <italic>r</italic> &#x0003D; &#x02212;0.831) and punishment trials (score: <italic>r</italic> &#x0003D; &#x02212;0.639, advantageous choices: <italic>r</italic> &#x0003D; &#x02212;0.870, all <italic>p</italic> &#x0003C; 0.005, CF 4). Importantly, both the number of switches before and after reaching the criterion was higher in the punishment compared to the reward condition (median and [25th, 75th percentile] values: reward (before) &#x0003D; 24.10 [9.09, 35,06]%, punishment (before) &#x0003D; 36.07 [25.57, 44.86]%, z &#x0003D; &#x02212;3.75, <italic>p</italic> &#x0003C; 0.001; reward (after) &#x0003D; 5.39 [0, 11.29]%, punishment (after) &#x0003D; 16.83 [8.02, 28.48]%, z &#x0003D; &#x02212;4.31, <italic>p</italic> &#x0003C; 0.0001, CF 2). Thus, the higher overall number of switches in the punishment condition was not restricted to exploring all choice options, but continued to be increased after successful learning. Note that because the number of trials before and after reaching the criterion varied across participants and conditions, we used the relative number of switches (in %) for this analysis.</p>
<p>Analyses of reaction times further supported our hypotheses. In line with hypothesis 1, learning was accompanied by a significant decrease in reaction times (RT) over the course of the experiment [main effect of task block: <italic>F</italic><sub>(2.035, 87.507)</sub> &#x0003D; 25.488, <italic>p</italic> &#x0003C; 0.001, all pairwise differences statistically significant with <italic>p</italic> &#x0003C; 0.0016, CF 6, except for the change from block 2 to block 3, Figure <xref ref-type="fig" rid="F2">2B</xref>]. In line with hypothesis 2, we observed a main effect of valence [<italic>F</italic><sub>(1.741, 74.855)</sub> &#x0003D; 36.866, <italic>p</italic> &#x0003C; 0.0001] with longest RTs in punishment trials (854.75 ms &#x000B1; 17.59), shortest RTs in reward trials (757.07 ms &#x000B1; 18.05) and RTs in neutral trials (811.56 ms &#x000B1; 17.51) in between (all pairwise comparisons statistically significant at <italic>p</italic> &#x0003C; 0.0001, CF 3).</p>
<p>Finally, changes in valence and arousal ratings before and after learning were in line with our behavioral findings, with significantly increased valence and arousal ratings for the high probability reward symbol, and significantly increased arousal ratings, but decreased valence ratings for the punishment symbols after learning (<xref ref-type="supplementary-material" rid="SM1">Supplementary Material</xref>). These findings corroborate the ecological validity of our task design.</p>
</sec>
<sec>
<title>Computational modeling and analysis of learning parameters</title>
<p>After fitting the model to each participant&#x00027;s behavioral data, we first accessed model adequacy by means of the consistency parameter &#x003B2;. Across participants, the model parameter &#x003B2; explained a significant 54% of the variability in switching behavior [linear regression, <italic>R</italic><sup>2</sup> &#x0003D; 0.54, adjusted <italic>R</italic><sup>2</sup> &#x0003D; 0.53, <italic>F</italic><sub>(1, 46)</sub> &#x0003D; 53.41, <italic>p</italic> &#x0003C; 0.0001] speaking for model behavior that, after model fitting, captured significant portions of variability in participants&#x00027; behavior. In addition, BIC-values obtained across participants did not significantly differ with respect to the factors gender and obesity (both <italic>p</italic> &#x0003E; 0.09). Thus, model fit was comparable across participant groups, a prerequisite for parameter comparison across groups as presented below.</p>
<p>From the fitted models, three independent learning rates for the reward, punishment, and neutral condition were derived for each subject. Across participants, learning rates in the reward (0.1 &#x000B1; 0.07) and the punishment (0.07 &#x000B1; 0.04) condition were significantly increased compared to the neutral [0.001, (0.001, 0.26)] condition (z &#x0003D; 3.687 and z &#x0003D; 3.551, respectively, both <italic>p</italic> &#x0003C; 0.0004, CF 2), again reflecting learning in both reinforcement-based conditions. In addition better learning performance in the reward compared to the punishment condition was accompanied in a small but significant difference in learning rates with a higher learning rate for reward compared to punishment trials [main effect of condition, <italic>F</italic><sub>(1, 47)</sub> &#x0003D; 4.09, <italic>p</italic> &#x0003D; 0.04]. This reflects the different speed of learning between the two conditions, as a smaller learning rate directly translates to a slower albeit correct update of value representations over the course of learning.</p>
</sec>
<sec>
<title>Impact of personality traits and working memory capacity</title>
<p>In order to ensure that none of the observed behavioral effects were simply attributable to a systematic impact of the previously identified personality traits or working memory, we first analyzed potential group differences of these factors and their correlations. Detailed statistical results of this analysis are provided as <xref ref-type="supplementary-material" rid="SM1">Supplementary Material</xref>. Small gender differences were observed for BAS reward responsiveness, total BIS score, UPPS urgency, UPPS perseverance, and the WMS-R score. Bivariate correlations between these scores and performance measures did not reach significance. Thus, none of the observed behavioral effects were merely reflecting differences in personality traits or working memory capacity. Consequently, we omitted these factors in the subsequent analysis of phasic IBI changes in response to the presentation of stimuli and feedback.</p>
</sec>
<sec>
<title>Analysis of autonomic responses</title>
<p>Sequences of mean IBIs from stimulus presentation to HR recovery after feedback presentation are shown in Figure <xref ref-type="fig" rid="F3">3</xref>. Across all participants, changes in IBI length are plotted separately for reward and punishment trials during the first and second experimental half. As becomes obvious by visual inspection already, in the first experimental half, IBI deceleration was stronger in the punishment compared to the reward condition for both the presentation of stimuli as well as feedback. These differences vanished later in the experiment. In order to disentangle the impact of stimulus and feedback presentation on IBI length, changes in IBI length were assessed in the following detailed statistical analyses independently for the presentation of stimuli and the presentation of feedback.</p>
<fig id="F3" position="float">
<label>Figure 3</label>
<caption><p>Phasic cardiac responses. Sequence of IBIs in response to reward (red) and punishment (blue) for the first <bold>(left)</bold> and the second <bold>(right)</bold> experimental half. For the purpose of plotting responses to stimuli and feedback on a common scale, in this figure all IBIs from stimulus presentation to HR recovery after feedback presentation are referenced to a common IBI &#x02212;2 prior to stimulus presentation and named in relation to stimulus presentation IBI 0 to IBI 7. Arrows mark the presentation of stimuli (ST) and feedback (FB). Note that the statistical analysis of IBIs was performed separately for stimulus and feedback presentation (see Figures <xref ref-type="fig" rid="F4">4</xref>, <xref ref-type="fig" rid="F5">5</xref>).</p></caption>
<graphic xlink:href="fnins-11-00598-g0003.tif"/>
</fig>
<p>First, we statistically analyzed mean IBI differences in response to <bold>stimulus</bold> presentation. In addition to the main effect of IBI [<italic>F</italic><sub>(1.68, 74.06)</sub> &#x0003D; 29.66, <italic>p</italic> &#x0003C; 0.0001], we observed a main effect of valence [<italic>F</italic><sub>(1, 44)</sub> &#x0003D; 12.44, <italic>p</italic> &#x0003D; 0.001] and a significant IBI &#x000D7; valence interaction [<italic>F</italic><sub>(2.38, 104.61)</sub> &#x0003D; 23.12, <italic>p</italic> &#x0003C; 0.0001, Figure <xref ref-type="fig" rid="F4">4A</xref>]. This was driven by a significantly higher increase in IBI length from IBI 0 to IBI 1 in punishment trials compared to reward trials [<italic>F</italic><sub>(1, 44)</sub> &#x0003D; 4.50, <italic>p</italic> &#x0003D; 0.040], representing stronger initial deceleration in response to stimuli predicting potential punishment. This was followed by a smaller decrease in IBI length from IBI 1 to IBI 2 in the punishment condition [<italic>F</italic><sub>(1, 44)</sub> &#x0003D; 27.01, <italic>p</italic> &#x0003C; 0.0001], representing a pronounced prolonged deceleration in response to stimuli predicting potential punishment. Thus, HR changes in response to stimulus presentation followed the pattern that we predicted in hypothesis 3 for HR responses to feedback, with pronounced reactivity for punishment compared to reward.</p>
<fig id="F4" position="float">
<label>Figure 4</label>
<caption><p>Cardiac responses to stimulus presentation. Effects of valence and time on changes in relative IBI length around stimulus presentation. <bold>(A)</bold> Deceleration in response to stimulus presentation was stronger and prolonged for stimuli predicting punishment compared to reward. <bold>(B)</bold> Cardiac reactivity in response to ST presentation was more pronounced during the second experimental half. Arrows mark the presentation of stimuli (ST). Significant differences (at <italic>p</italic> &#x0003C; 0.05) between conditions or experimental half 1 and 2 in the slope between neighboring IBIs are marked with <sup>&#x0002A;</sup>.</p></caption>
<graphic xlink:href="fnins-11-00598-g0004.tif"/>
</fig>
<p>Additionally, we observed a main effect of experimental half [<italic>F</italic><sub>(1, 44)</sub> &#x0003D; 10.30, <italic>p</italic> &#x0003D; 0.002] and a significant IBI &#x000D7; experimental half interaction [<italic>F</italic><sub>(2.3, 100.99)</sub> &#x0003D; 5.50, <italic>p</italic> &#x0003D; 0.004, Figure <xref ref-type="fig" rid="F4">4B</xref>]. This interaction resulted from higher increase in IBI length from IBI &#x02212;1 to IBI 0 [<italic>F</italic><sub>(1, 44)</sub> &#x0003D; 10.86, <italic>p</italic> &#x0003D; 0.002] and a stronger decrease from IBI 1 to IBI 2 [<italic>F</italic><sub>(1, 44)</sub> &#x0003D; 9.57, <italic>p</italic> &#x0003D; 0.003] in the second compared to the first half of the experiment. Thus, across reward and punishment conditions, anticipatory deceleration to the stimulus and recovery after stimulus presentation increased significantly over the course of the experiment. This already points at a shift in HR responsiveness over time, which is more directly addressed below.</p>
<p>Second, we analyzed mean IBI differences in response to <bold>feedback</bold>. The analysis revealed a main effect of IBI [<italic>F</italic><sub>(2.05, 89.97)</sub> &#x0003D; 9.07, <italic>p</italic> &#x0003C; 0.0001] and a significant three-way interaction of experimental half &#x000D7; IBI &#x000D7; valence [<italic>F</italic><sub>(2.56, 112.51)</sub> &#x0003D; 4.453, <italic>p</italic> &#x0003D; 0.008], again signifying an impact of the factors valence and time on IBI. However, as we also found a significant four-way interaction experimental half &#x000D7; IBI &#x000D7; valence &#x000D7; gender [<italic>F</italic><sub>(2.56, 112.51)</sub> &#x0003D; 2.90, <italic>p</italic> &#x0003D; 0.046], the impact of valence and time cannot be interpreted without considering the factor gender.</p>
<p>Regarding the factor valence, our analysis shows that IBI 0, IBI 1, IBI 2 were significantly longer in the punishment compared to the reward condition, but only in women (<italic>F</italic><sub>(1, 44)</sub> &#x0003D; 6.157, <italic>p</italic> &#x0003D; 0.017; <italic>F</italic><sub>(1, 44)</sub> &#x0003D; 8.931, <italic>p</italic> &#x0003D; 0.005; <italic>F</italic><sub>(1, 44)</sub> &#x0003D; 4.253, <italic>p</italic> &#x0003D; 0.045, respectively, Figure <xref ref-type="fig" rid="F5">5</xref> red lines), not in men (all <italic>p</italic> &#x0003E; 0.66, Figure <xref ref-type="fig" rid="F5">5</xref> green lines). The difference between reward and punishment in women was driven by higher anticipatory deceleration from IBI &#x02212;1 to IBI 0 in punishment compared to reward trials [<italic>t</italic><sub>(23)</sub> &#x0003D; 2.28, <italic>p</italic> &#x0003D; 0.033], and a prolonged deceleration from IBI 2 to IBI 3 in the reward compared to the punishment condition [<italic>t</italic><sub>(23)</sub> &#x0003D; 4.97, <italic>p</italic> &#x0003C; 0.0001]. Thus, while these findings support our hypothesis 3 of a significant HR deceleration in response to negative feedback, the expected differential effect of positive and negative feedback was gender specific. Importantly, the effects of feedback valence on IBI were significant only in the first, but not in the second experimental half, which again points at a shift in HR responsiveness over time.</p>
<fig id="F5" position="float">
<label>Figure 5</label>
<caption><p>Effects of weight status and gender. Interaction between gender and valence on changes in relative IBI length around feedback presentation during the first experimental half. The interaction was driven by (1) stronger overall cardiac reactivity to feedback presentation in women compared to men (red vs. green lines), and (2) stronger anticipatory deceleration and faster recovery in the punishment <bold>(B)</bold> compared to the reward condition <bold>(A)</bold> in women only with no observable differences in men. None of these effects was observable in the second half of the experiment. Arrows mark the presentation of feedback (FB). Significant differences (at <italic>p</italic> &#x0003C; 0.05) between genders in the slope between neighboring IBIs are marked with <sup>&#x0002A;</sup>.</p></caption>
<graphic xlink:href="fnins-11-00598-g0005.tif"/>
</fig>
<p>To directly address the hypothesized <bold>shift in HR responsiveness</bold> from feedback to stimulus presentation during learning, we analyzed for each subject the mean area under the curve (AUC) of changes in IBI length following stimulus and feedback presentation. Most importantly, we observed an event type (stimulus/feedback presentation) &#x000D7; experimental half interaction [<italic>F</italic><sub>(1, 44)</sub> &#x0003D; 12.97, <italic>p</italic> &#x0003D; 0.001] which was driven by significantly higher responses to stimulus presentation compared to feedback presentation in the second half of the experiment [<italic>F</italic><sub>(1, 44)</sub> &#x0003D; 8.70, <italic>p</italic> &#x0003D; 0.005]. This strongly supports our hypothesis 4 as it represents the expected shift from relying on external feedback to update choice behavior to an internal monitoring system evaluating acquired knowledge in the course of learning.</p>
<p>Addressing our final hypothesis, we assessed the relationship between IBI length and strength of PE signal after the presentation of positive or negative feedback. This relationship was modeled by a GLM with subsequent statistical analysis of the GLM coefficients. Across subjects, all GLM coefficients corresponding to IBI 0, IBI 1, and IBI 2 after feedback presentation significantly differed from zero [all <italic>t</italic><sub>(47)</sub> &#x0003E; 5.89, all <italic>p</italic> &#x0003C; 0.001, CF 6]. Thus, across subjects the three IBIs following positive or negative feedback were systematically correlated with the strength of the PE signal.</p>
<p>Comparing standardized coefficients across IBIs and conditions revealed a main effect of valence [<italic>F</italic><sub>(1, 47)</sub> &#x0003D; 6.21, <italic>p</italic> &#x0003D; 0.016] with higher mean coefficients in the punishment (0.516 &#x000B1; 0.055) compared to the reward condition (0.386 &#x000B1; 0.036), and a main effect of IBI [<italic>F</italic><sub>(2, 94)</sub> &#x0003D; 8.28, <italic>p</italic> &#x0003C; 0.001] with a significantly higher coefficient for IBI 1 (0.552 &#x000B1; 0.060) compared to IBI 0 (0.369 &#x000B1; 0.0034) and compared to IBI 2 (0.432 &#x000B1; 0.042, <italic>p</italic> &#x0003D; 0.0003 and <italic>p</italic> &#x0003D; 0.008, respectively, CF 3). In line with our hypothesis, these results show that phasic changes in IBI following feedback processing, in particular changes in IBI 1, directly reflect the strength of PE signals, i.e., the degree of discrepancy between expected and actual outcome of an action. This relationship is particularly pronounced for negative feedback.</p>
</sec>
<sec>
<title>The effects of weight status and gender</title>
<p>Weight status and gender influenced some but not all investigated aspects of reinforcement learning and feedback processing both on the behavioral and the physiological level. For the sake of succinctness, we summarize below all significant effects of these two factors in the different analyses including corresponding <italic>p</italic>-values. Detailed statistical analyses of these effects can be found in the <xref ref-type="supplementary-material" rid="SM1">Supplementary Material</xref>.</p>
<p>On the <bold>behavioral level</bold>, we observed a significant condition-specific impact of weight status on learning speed. In the punishment condition, participants with obesity reached the learning criterion significantly later than lean participants (<italic>p</italic> &#x0003D; 0.036). With respect to the factor gender, we observed a difference in task performance with higher overall scores for men than in women (at a trend level <italic>p</italic> &#x0003D; 0.094) and more advantageous choices for men than women in the reward condition (<italic>p</italic> &#x0003D; 0.036, Figure <xref ref-type="fig" rid="F2">2C</xref>). This was accompanied by a significantly higher number of switches in women than men in both reward (<italic>p</italic> &#x0003D; 0.023) and punishment trials (<italic>p</italic> &#x0003D; 0.045, Figure <xref ref-type="fig" rid="F2">2D</xref>). Most importantly, in the reward condition women more often than men continued to switch between choices <italic>after</italic> reaching the learning criterion (<italic>p</italic> &#x0003D; 0.007), leading to reduced performance in learning from reward in women. In contrast, learning rates derived from the computational model did not differ between genders (<italic>p</italic> &#x0003D; 0.80). Thus, the observed performance differences between genders were not rooted in differential integration of new experiences into existing knowledge, but rather in the inconsistency of choice behavior as reflected in the increased switching in women even after successful learning.</p>
<p>On the <bold>autonomic level</bold>, we observed a three-way interaction of IBI with obesity and gender in the phasic cardiac responses to stimulus presentation (<italic>p</italic> &#x0003D; 0.039). This interaction was driven by an increased initial deceleration in response to stimulus presentation in lean men. We further observed gender differences in cardiac responses to positive and negative feedback presentation during the first experimental half. In reward trials, women showed slower HR recovery than men with smaller HR deceleration from IBI 1 to IBI 2 and from IBI 2 to IBI 3 (<italic>p</italic> &#x0003D; 0.015 and <italic>p</italic> &#x0003D; 0.024, respectively, Figure <xref ref-type="fig" rid="F5">5A</xref>). In the punishment condition, women showed overall increased HR responses compared to men, caused by a stronger anticipatory deceleration from IBI &#x02212;1 to IBI 0 (<italic>p</italic> &#x0003D; 0.006, Figure <xref ref-type="fig" rid="F5">5B</xref>).</p>
<p>Finally, a three-way interaction of experimental half, gender, and obesity was observed in AUC-values (<italic>p</italic> &#x0003D; 0.024) with significantly higher responses in lean men compared to lean women during the first experimental half (<italic>p</italic> &#x0003D; 0.022). This speaks for a faster internalization of stimulus-outcome associations in lean men during the initial phase of learning.</p>
</sec>
</sec>
<sec sec-type="discussion" id="s4">
<title>Discussion</title>
<p>Successful learning and behavioral adaptation hinges on the sufficient detection and adequate evaluation of external feedback and several studies have established a link between the processing of external feedback and autonomic reactions (e.g., Somsen et al., <xref ref-type="bibr" rid="B86">2000</xref>; Crone et al., <xref ref-type="bibr" rid="B19">2003</xref>, <xref ref-type="bibr" rid="B16">2005</xref>; Groen et al., <xref ref-type="bibr" rid="B39">2007</xref>). Using a probabilistic learning task, we investigated the cardiac concomitants of reinforcement-based learning and the impact of weight status and gender on learning performance. Further, we introduced a new method for simultaneously analyzing behavioral and autonomic data that enabled us to link these two modalities on a trial-by-trial basis. Our study makes several important contributions to our understanding of reinforcement learning and related autonomic reactions. We could show that learning and feedback processing is closely mirrored by phasic cardiac responses on several levels (1) On a trial-by-trial basis phasic cardiac responses after feedback are correlated with the strength of PE signals that encode the deviation between expected and actual outcome of a choice or action. (2) Cardiac responses shifted from feedback presentation at the beginning of learning to stimulus presentation at later stages. (3) Feedback valence impacted on cardiac responses with faster and prolonged HR deceleration in response to negative feedback. Additionally, we observed differential impacts of weight-status and gender on both learning performance and changes in HR responses. In the following, we discuss these results in more detail.</p>
<p>Several previous studies have shown that during reinforcement-based learning, the processing of feedback is reflected in HR slowing. However, investigations into the precise meaning of the observed effects yielded heterogeneous results so far. Van Der Veen et al. (<xref ref-type="bibr" rid="B93">2004</xref>) reported that cardiac slowing was stronger and prolonged for negative compared to positive feedback, but did not discriminate between informative and non-informative feedback. They argued that HR deceleration may thus be sensitive to the valence rather than relevance of feedback. Others found that cardiac responses were stronger toward unexpected feedback and suggested that this reflects the monitoring of learning-relevant information (Somsen et al., <xref ref-type="bibr" rid="B86">2000</xref>; Crone et al., <xref ref-type="bibr" rid="B19">2003</xref>, <xref ref-type="bibr" rid="B16">2005</xref>). With our approach we were able to directly address these different views, linking learning performance and behavioral adaptation to estimates of internal learning signals. If cardiac responses merely reflected feedback valence, no direct link to PE signals would be expected. In line with previous studies, our results show an overall stronger and prolonged HR deceleration in response to punishment compared to reward. However, we additionally found that the strength of HR deceleration following feedback was indeed predictive of the strength of the model-derived PEs that indicate how much the provided feedback deviated from the participants&#x00027; expectations. Interestingly, this relationship was particularly pronounced for negative feedback, which signals the need for behavioral adjustments, while positive feedback reinsures the learner that the current choice behavior is correct.</p>
<p>Supporting our hypotheses, we further observed that HR responses toward feedback changed over the course of learning. Specifically, each symbol pair in the current task was exclusively associated with the prospect of a reward, threat of punishment, or financially neutral feedback and these associations did not change over time. Consequently, after an initial learning phase the participants should have been able to anticipate the potential trial outcome associated with a presented symbol pair. Indeed, this was reflected in HR responses, showing a shift of cardiac responses from the presentation of feedback during the first half of the experiment to the presentation of the symbol pairs in the second half. This corroborates findings by Groen et al. (<xref ref-type="bibr" rid="B39">2007</xref>), who observed a general reduction in feedback-related HR deceleration with learning, together with a shift of HR slowing from the IBI following feedback presentation to the IBI preceding feedback presentation in later stages of the experiment.</p>
<p>Taken together, these results provide strong new evidence for the assumption that HR deceleration during learning is sensitive to learning-relevant information and reflects an internal monitoring system to detect the violation of expectations derived from preceding experience, while over the course of learning a shift from the dependency on external feedback signals at initial stages to the use of internal error detection mechanisms occurs.</p>
<p>Successful feedback-based learning requires the integration of multiple processes. In each trial of a reinforcement learning paradigm, the learner needs to monitor incoming feedback, build and maintain value representations, construct a PE signal, update value representations according to the PE and, when necessary, adjust behavior for future actions and decisions. Deficits in feedback-based learning can be caused by impairments in any of these sub-processes or a combination thereof. For example, phenomenologically similar deficits in reinforcement learning can be observed in young children and older adults, but these deficits are likely attributable to impairments in different underlying mechanisms: a reduced executive control capacity in children, and a decline in the ability for differentiated value representation with age (H&#x000E4;mmerer and Eppinger, <xref ref-type="bibr" rid="B41">2012</xref>). However, based on behavioral observations alone, the various aspects of the learning process cannot clearly be disentangled.</p>
<p>Our behavioral data combined with computational modeling allowed us to partly disentangle the sub-processes of reinforcement learning in a within-subject fashion. In line with our hypothesis, learning was evident on the behavioral level from increasing overall task scores, increasing numbers of optimal choices, and decreasing reaction times over time across participants. In addition, participants changed their valence and arousal ratings for the symbols according to their probabilities of predicting reward or punishment. Together, these results signify behavioral adaptations that were generally appropriate for the task at hand. Further, higher model-derived learning rates in the reward and punishment compared to the neutral condition point at appropriate updating of value representations in both conditions. However, we also observed systematic differences between the processing of positive and negative feedback with more advantageous choices and shorter reaction times for reward than for punishment. Computational modeling revealed a small difference in learning rate between the reward and punishment condition. This was corroborated by the fact that participants on average needed longer to reach the learning criterion in the punishment than in the reward condition. Both results indicate a slower updating of value representations after negative feedback. In addition, reduced performance in the punishment condition could be linked to switching behavior with more switching after negative than after positive feedback. While this speaks for an appropriate behavioral adaptation after punishment, increased switching was also observable after the learning criterion was reached, i.e., after participants should have learned that it is advantageous to stick to a certain symbol, even if it is occasionally punished.</p>
<p>In sum, behavioral analysis and computational modeling suggest that the observed differences in task performance when learning from reward and learning from punishment were not caused by insufficient sensitivity to or internal representation of negative feedback. Rather they are attributable to (1) a difference in value updating or, in other words, different speed of learning between these conditions and (2) differences in behavioral adaptation in the exploration and exploitation phase of learning with continued increased switching after successful learning in the punishment condition. While our analyses cannot provide a full explanation of differences in positive and negative feedback, they would predict similar value representations and PE signals on the neural level, while the utilization of these signals for learning might differ. This hypothesis will be subject of our future investigations.</p>
<p>In addition to general psychophysiological correlates of reinforcement-based learning, we hypothesized that weight status and gender might impact on performance and HR responses for rewards and punishments. Supporting our hypotheses, we found that individuals with obesity showed a slower learning of advantageous choice behavior in the punishment learning condition. Specifically, they needed more time to learn to stably choose the advantageous choice option. This is in line with previous reports of a compromised learning performance in individuals with obesity (Coppin et al., <xref ref-type="bibr" rid="B15">2014</xref>). Interestingly, in the current study, weight status was not related to other performance measures that captured behavior across the whole experiment (e.g., learning rate, number of advantageous choices). This suggests that differences were restricted to the initial learning phase, while individuals with obesity were able to compensate and reach a comparable performance across the whole experiment. Indeed, using the same paradigm in a functional magnetic resonance study, we found obesity-related impairments particularly during the first half of the experiment (Kube et al., submitted). In the same vein, Zhang et al. (<xref ref-type="bibr" rid="B105">2014</xref>) reported differences in learning performance between lean and obese women within as few as &#x0007E;20 trials, supporting the idea of an early acquisition deficit in obesity. Various mechanisms for impaired reinforcement-based learning have been identified in other populations, showing alterations in PE encoding in aging (Eppinger et al., <xref ref-type="bibr" rid="B24">2013</xref>), PE utilization in addiction (Park et al., <xref ref-type="bibr" rid="B71">2010</xref>), and working memory capacity in healthy individuals (Collins and Frank, <xref ref-type="bibr" rid="B13">2012</xref>) to be related to a poorer performance. Indeed, in a recent paper, Collins et al. (<xref ref-type="bibr" rid="B14">2017</xref>) argue that learning in simple reinforcement-based tasks is best explained by a mixture of working memory and PE processes. Though we have not found group differences in visual working memory in the current study, obesity-related impairments in other measures of working memory capacity have been shown to affect preference learning (Coppin et al., <xref ref-type="bibr" rid="B15">2014</xref>). Additionally, we have previously linked reinforcement learning deficits in obesity to a less efficient utilization of negative PEs in the striatum (Mathar et al., <xref ref-type="bibr" rid="B63">2017a</xref>), thus adding another potential mechanism to explain obesity-related effects in the current study. In sum, these results suggest that individuals with obesity exhibit a slower learning performance. However, so far, the underlying mechanisms have not been fully discovered, with potentially different mechanisms interacting to explain learning alterations in individuals with obesity.</p>
<p>Complementary to the effect of weight-status, we found a modulation of learning performance and HR responses by gender. In line with numerous previous studies (e.g., Crone et al., <xref ref-type="bibr" rid="B19">2003</xref>; Groen et al., <xref ref-type="bibr" rid="B39">2007</xref>), cardiac responses to external stimuli were characterized by an initial HR deceleration, followed by an acceleratory recovery response. However, in our study, women compared to men exhibited a prolonged HR deceleration after rewards and a stronger initial deceleration to punishment particularly during the first half of the experiment. As detailed above, previous studies have mostly reported a stronger and prolonged deceleration to the presentation of negative stimuli (Crone et al., <xref ref-type="bibr" rid="B17">2004a</xref>; Van Der Veen et al., <xref ref-type="bibr" rid="B93">2004</xref>), but in the context of learning stronger HR responses may likewise be associated to the processing of learning-relevant information in general (Somsen et al., <xref ref-type="bibr" rid="B86">2000</xref>; Crone et al., <xref ref-type="bibr" rid="B19">2003</xref>, <xref ref-type="bibr" rid="B16">2005</xref>). Further, responses may also depend on the motivational significance of the stimulus material. For instance, a stronger deceleration seems to occur for large compared to small monetary losses (Crone et al., <xref ref-type="bibr" rid="B18">2004b</xref>), while highly arousing positive and negative pictures have been found to elicit stronger cardiac deceleration than low arousing emotional stimuli (Balconi et al., <xref ref-type="bibr" rid="B2">2009</xref>). Consequently, stronger HR responses to both reward and punishment in women compared to men could speak for a stronger utilization of learning-relevant information or a generally heightened sensitivity to feedback stimuli in women.</p>
<p>Surprisingly, this was not directly mirrored in the learning indices, as women exhibited a poorer performance than men, particularly when learning from reward feedback. Gender-related influences on performance in reward-based choice tasks have been frequently reported in other studies. For instance, men have been shown to exhibit a higher performance in reversal learning tasks than women (Evans and Hampson, <xref ref-type="bibr" rid="B26">2015</xref>) and likewise outperform women in the Iowa Gambling task (Weller et al., <xref ref-type="bibr" rid="B99">2009</xref>; Evans and Hampson, <xref ref-type="bibr" rid="B26">2015</xref>). Interestingly, this seems to be driven by the fact that men quickly learn to choose cards from decks associated with smaller immediate rewards, but a larger net payoff across trials, while women keep choosing from a deck with frequent high immediate rewards and even higher, but infrequent losses (Overman, <xref ref-type="bibr" rid="B70">2004</xref>). Males thus appear to be more sensitive to the long-term monetary outcomes of the task, and females are more sensitive to immediate rewards. Indeed, in our study, women were characterized by higher trait reward sensitivity than men. Likewise, stronger HR responses to feedback may be an indicator of heightened feedback sensitivity in women. Interestingly, an allegedly lower learning performance in women than men was accompanied by comparable learning rates, speaking for similar value updating processes in both genders and against insufficient feedback monitoring in women, respectively. Instead, women showed more inconsistent choice behavior, i.e., even after they had stably learned to choose the more advantageous symbol in reward trials, they more often switched to the other (disadvantageous) symbol than men. In the light of previous studies, this could indicate that despite their knowledge of the advantageous choice options, women were more susceptible to the presentation of probabilistic (misleading) feedback, i.e., the infrequent omission of an expected reward after an advantageous choice and the infrequent receipt of a reward after a disadvantageous choice may have fostered switching behavior more strongly in women than men. In sum, our results suggest that the observed performance deficits in learning from reward in women were not caused by deficits in feedback monitoring or the representation and updating of stimulus values, but by a higher responsiveness to reinforcement that was accompanied by more pronounced HR responses and interfered with the beneficial behavior in the current study.</p>
<p>Lastly, we found evidence for a combined influence of obesity and gender on HR responses. Specifically, lean men exhibited a stronger initial deceleration during stimulus presentation than lean women, suggesting a stronger anticipatory response to the prospect of reinforcement. However, HR changes did not translate to alterations in behavioral performance and no differences were found between men and women with obesity. This is a clearly surprising finding, especially, since previous studies have highlighted that alterations in executive functioning and behavioral adaptation may be particularly pronounced in women with obesity, while performance of men with obesity remains relatively intact (Weller et al., <xref ref-type="bibr" rid="B100">2008</xref>; Horstmann et al., <xref ref-type="bibr" rid="B44">2011</xref>; Zhang et al., <xref ref-type="bibr" rid="B105">2014</xref>). Consequently, the influence of gender on obesity-effects seems specific for certain types of stimuli and tasks and requires further consolidation.</p>
<p>Finally, some limitations of the current study design must be acknowledged: First, it has been shown that cardiac markers are significantly influenced by stimulus timing during cognitive and emotional processing. For instance, negative emotional stimuli presented at systole are detected more easily (Garfinkel et al., <xref ref-type="bibr" rid="B34">2014</xref>) and perceived as more intense than stimuli presented at diastole (Gray et al., <xref ref-type="bibr" rid="B38">2012</xref>; Garfinkel et al., <xref ref-type="bibr" rid="B34">2014</xref>), while words encoded at systole are less well remembered than words encoded at diastole (Garfinkel et al., <xref ref-type="bibr" rid="B33">2013</xref>). In the current study, stimulus and feedback presentation were not time-locked to the onset of systole or diastole. Instead they were presented at variable points within the cardiac cycle. This leaves the possibility that differences in cardiac responses may have been affected by incidental differences in stimulus timing within the cardiac cycle. However, trial order was pseudo-randomized and trials were separated by varying ITI lengths and delay periods, thus rendering an influence of stimulus onset unlikely. Second, the interval between stimulus and feedback presentation was relatively small with a maximum of 3,500 ms. While this is sufficient to minimize the impact of the strongest autonomic reactions to the stimulus (IBI 0 and IBI 1) on the following feedback presentation, a full recovery of cardiac responses before feedback presentation is unlikely. It would clearly be ideal to separate both phases by longer delay periods to await a recovery to baseline before feedback presentation. However, this would in turn result in significantly longer trials and a significantly increased duration of the experiment, which can potentially facilitate fatigue and decreases attention toward the task. Instead, we used an IBI <italic>after</italic> stimulus presentation as reference for the feedback-related IBIs, thus technically excluding any stimulus-related carryover effects from stimulus to feedback presentation. Similarly, feedback was followed by the next trial&#x00027;s stimulus presentation after 3,600 to 4,200 ms which again might not have been entirely sufficient for full recovery. To alleviate this problem, we used jittered ITIs and a pseudo-randomized trial order, and statistically ensured that the reference IBIs for the stimulus analyses were independent of all factors that could impact on the subsequent IBIs. Third, we did not measure respiration in the current study, though it has been shown that HR fluctuates depending on respiration. Heart periods become shorter or longer in phase relationship with inspiration and expiration (Berntson et al., <xref ref-type="bibr" rid="B5">1993</xref>). While some highlight the need to remove respiratory influences from the ECG signal (Quintana and Heathers, <xref ref-type="bibr" rid="B73">2014</xref>), others argue that resting HR and respiration share a common basis (Thayer et al., <xref ref-type="bibr" rid="B91">2011</xref>). Thus, under spontaneous breathing conditions, controlling for respiration may remove variance in the ECG signal that the researcher is actually interested in (Laborde et al., <xref ref-type="bibr" rid="B56">2017</xref>). Nevertheless, measuring respiration simultaneously to ECG could have helped to detect non-cyclical breathing patterns (e.g. sighs) that could bias HR results (Vaschillo et al., <xref ref-type="bibr" rid="B94">2015</xref>). Lastly, it would have been interesting to investigate in more detail a potential link between the stimulus&#x02014;and feedback-related cardiac responses during learning with the observed differences in switching behavior in the reward and punishment condition and between genders, in particular after reaching the learning criterion. Unfortunately, our task design did not allow for such a detailed analysis as the number of switch trials after successful learning were too small for a statistically sound analysis. An experimental design that provokes higher switching rates and includes conditions where switching might also be an advantageous strategy together with ECG measurements could be an interesting approach for future work to answer this question.</p>
</sec>
<sec sec-type="conclusions" id="s5">
<title>Conclusion</title>
<p>In the present study, we investigated learning performance and cardiac concomitants of reinforcement learning together with the impact of feedback valence, gender, and weight status on learning performance and autonomic responses. We could show that the strength of cardiac responses to learning-related feedback directly reflects the strength of PE signals that alert the learner to the necessity for value updating and behavioral adaptation. Thus, phasic changes in HR seem to express processes of an internal feedback monitoring system that is sensitive to the violation of performance-based expectations. Moreover, apparent gender-related deficits in reinforcement learning were not caused by deficiencies in knowledge acquisition, but by insufficient adaptation in an environment that requires consistent choice behavior. Finally, our study adds evidence to the notion that individuals with obesity might be impaired in learning to avoid negative outcomes.</p>
</sec>
<sec id="s6">
<title>Author contributions</title>
<p>LK, JK, AV, and JN conceived of the study and designed it. LK and JK performed the measurements. LK, JK, and JN analyzed the data. LK, JK, AV, and JN wrote the manuscript.</p>
<sec>
<title>Conflict of interest statement</title>
<p>The authors declare that the research was conducted in the absence of any commercial or financial relationships that could be construed as a potential conflict of interest.</p></sec>
</sec>
</body>
<back>
<ack><p>We thank all participants involved in this study for their cooperation, Ramona Menger and Bettina Johst for their assistance in programming the paradigm and recruitment, Annette Horstmann, Isabel Garc&#x000ED;a-Garc&#x000ED;a, David Mathar, Kathleen Wiencke, and Christoph M&#x000FC;hlberg for valuable discussions.</p>
</ack>
<sec sec-type="supplementary-material" id="s7">
<title>Supplementary material</title>
<p>The Supplementary Material for this article can be found online at: <ext-link ext-link-type="uri" xlink:href="https://www.frontiersin.org/articles/10.3389/fnins.2017.00598/full#supplementary-material">https://www.frontiersin.org/articles/10.3389/fnins.2017.00598/full#supplementary-material</ext-link></p>
<supplementary-material xlink:href="DataSheet1.docx" id="SM1" mimetype="application/vnd.openxmlformats-officedocument.wordprocessingml.document" xmlns:xlink="http://www.w3.org/1999/xlink"/>
</sec>
<ref-list>
<title>References</title>
<ref id="B1">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Alonso-Alonso</surname> <given-names>M.</given-names></name> <name><surname>Woods</surname> <given-names>S. C.</given-names></name> <name><surname>Pelchat</surname> <given-names>M.</given-names></name> <name><surname>Grigson</surname> <given-names>P. S.</given-names></name> <name><surname>Stice</surname> <given-names>E.</given-names></name> <name><surname>Farooqi</surname> <given-names>S.</given-names></name> <etal/></person-group>. (<year>2015</year>). <article-title>Food reward system: current perspectives and future research needs</article-title>. <source>Nutr. Rev.</source> <volume>73</volume>, <fpage>296</fpage>&#x02013;<lpage>307</lpage>. <pub-id pub-id-type="doi">10.1093/nutrit/nuv002</pub-id><pub-id pub-id-type="pmid">26011903</pub-id></citation></ref>
<ref id="B2">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Balconi</surname> <given-names>M.</given-names></name> <name><surname>Brambilla</surname> <given-names>E.</given-names></name> <name><surname>Falbo</surname> <given-names>L.</given-names></name></person-group> (<year>2009</year>). <article-title>Appetitive vs. defensive responses to emotional cues. Autonomic measures and brain oscillation modulation</article-title>. <source>Brain Res.</source> <volume>1296</volume>, <fpage>72</fpage>&#x02013;<lpage>84</lpage>. <pub-id pub-id-type="doi">10.1016/j.brainres.2009.08.056</pub-id><pub-id pub-id-type="pmid">19703433</pub-id></citation></ref>
<ref id="B3">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Balodis</surname> <given-names>I. M.</given-names></name> <name><surname>Kober</surname> <given-names>H.</given-names></name> <name><surname>Worhunsky</surname> <given-names>P. D.</given-names></name> <name><surname>White</surname> <given-names>M. A.</given-names></name> <name><surname>Stevens</surname> <given-names>M. C.</given-names></name> <name><surname>Pearlson</surname> <given-names>G. D.</given-names></name> <etal/></person-group>. (<year>2013</year>). <article-title>Monetary reward processing in obese individuals with and without binge eating disorder</article-title>. <source>Biol. Psychiatry</source> <volume>73</volume>, <fpage>877</fpage>&#x02013;<lpage>886</lpage>. <pub-id pub-id-type="doi">10.1016/j.biopsych.2013.01.014</pub-id><pub-id pub-id-type="pmid">23462319</pub-id></citation></ref>
<ref id="B4">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Beck</surname> <given-names>A.</given-names></name> <name><surname>Steer</surname> <given-names>R. A.</given-names></name></person-group> (<year>1993</year>). <source>Beck Depression Inventory Manual</source>. <publisher-loc>San Antonio, TX</publisher-loc>: <publisher-name>Psychological Corporation</publisher-name>.</citation></ref>
<ref id="B5">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Berntson</surname> <given-names>G. G.</given-names></name> <name><surname>Cacioppo</surname> <given-names>J. T.</given-names></name> <name><surname>Quigley</surname> <given-names>K. S.</given-names></name></person-group> (<year>1993</year>). <article-title>Respiratory sinus arrhythmia - autonomic origins, physiological-mechanisms, and psychophysiological implications</article-title>. <source>Psychophysiology</source> <volume>30</volume>, <fpage>183</fpage>&#x02013;<lpage>196</lpage>. <pub-id pub-id-type="doi">10.1111/j.1469-8986.1993.tb01731.x</pub-id><pub-id pub-id-type="pmid">8434081</pub-id></citation></ref>
<ref id="B6">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>B&#x000F3;di</surname> <given-names>N.</given-names></name> <name><surname>K&#x000E9;ri</surname> <given-names>S.</given-names></name> <name><surname>Nagy</surname> <given-names>H.</given-names></name> <name><surname>Moustafa</surname> <given-names>A.</given-names></name> <name><surname>Myers</surname> <given-names>C. E.</given-names></name> <name><surname>Daw</surname> <given-names>N.</given-names></name> <etal/></person-group>. (<year>2009</year>). <article-title>Reward-learning and the novelty-seeking personality: a between- and within-subjects study of the effects of dopamine agonists on young Parkinson&#x00027;s patients</article-title>. <source>Brain</source> <volume>132</volume>, <fpage>2385</fpage>&#x02013;<lpage>2395</lpage>. <pub-id pub-id-type="doi">10.1093/brain/awp094</pub-id><pub-id pub-id-type="pmid">19416950</pub-id></citation></ref>
<ref id="B7">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bradley</surname> <given-names>M. M.</given-names></name> <name><surname>Lang</surname> <given-names>P. J.</given-names></name></person-group> (<year>1994</year>). <article-title>Measuring emotion: the self-assessment manikin and the semantic differential</article-title>. <source>J. Behav. Ther. Exp. Psychiatry</source> <volume>25</volume>, <fpage>49</fpage>&#x02013;<lpage>59</lpage>. <pub-id pub-id-type="doi">10.1016/0005-7916(94)90063-9</pub-id><pub-id pub-id-type="pmid">7962581</pub-id></citation></ref>
<ref id="B8">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Bratusch-Marrain</surname> <given-names>P.</given-names></name> <name><surname>Schmid</surname> <given-names>P.</given-names></name> <name><surname>Waldh&#x000E4;usl</surname> <given-names>W.</given-names></name> <name><surname>Schlick</surname> <given-names>W.</given-names></name></person-group> (<year>1978</year>). <article-title>Specific weight loss in hyperthyroidism</article-title>. <source>Horm. Metab. Res.</source> <volume>10</volume>, <fpage>412</fpage>&#x02013;<lpage>415</lpage>. <pub-id pub-id-type="doi">10.1055/s-0028-1093403</pub-id><pub-id pub-id-type="pmid">711136</pub-id></citation></ref>
<ref id="B9">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Brinkmann</surname> <given-names>K.</given-names></name> <name><surname>Franzen</surname> <given-names>J.</given-names></name></person-group> (<year>2013</year>). <article-title>Not everyone&#x00027;s heart contracts to reward: insensitivity to varying levels of reward in dysphoria</article-title>. <source>Biol. Psychol.</source> <volume>94</volume>, <fpage>263</fpage>&#x02013;<lpage>271</lpage>. <pub-id pub-id-type="doi">10.1016/j.biopsycho.2013.07.003</pub-id><pub-id pub-id-type="pmid">23872164</pub-id></citation></ref>
<ref id="B10">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Brinkmann</surname> <given-names>K.</given-names></name> <name><surname>Franzen</surname> <given-names>J.</given-names></name></person-group> (<year>2017</year>). <article-title>Blunted cardiovascular reactivity during social reward anticipation in subclinical depression</article-title>. <source>Int. J. Psychophysiol</source>. <volume>119</volume>, <fpage>119</fpage>&#x02013;<lpage>126</lpage>. <pub-id pub-id-type="doi">10.1016/j.ijpsycho.2017.01.010</pub-id><pub-id pub-id-type="pmid">28130127</pub-id></citation></ref>
<ref id="B11">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Cacciatori</surname> <given-names>V.</given-names></name> <name><surname>Gemma</surname> <given-names>M. L.</given-names></name> <name><surname>Bellavere</surname> <given-names>F.</given-names></name> <name><surname>Castello</surname> <given-names>R.</given-names></name> <name><surname>De Gregori</surname> <given-names>M. E.</given-names></name> <name><surname>Zoppini</surname> <given-names>G.</given-names></name> <etal/></person-group>. (<year>2000</year>). <article-title>Power spectral analysis of heart rate in hypothyroidism</article-title>. <source>Eur. J. Endocrinol.</source> <volume>143</volume>, <fpage>327</fpage>&#x02013;<lpage>333</lpage>. <pub-id pub-id-type="doi">10.1530/eje.0.1430327</pub-id><pub-id pub-id-type="pmid">11022173</pub-id></citation></ref>
<ref id="B12">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Carver</surname> <given-names>C. S.</given-names></name> <name><surname>White</surname> <given-names>T. L.</given-names></name></person-group> (<year>1994</year>). <article-title>Behavioral inhibition, behaviorral activation, and affective responses to impending reward and punishment: the BIS/BAS scales</article-title>. <source>J. Pers. Soc. Psychol.</source> <volume>67</volume>, <fpage>319</fpage>&#x02013;<lpage>333</lpage>. <pub-id pub-id-type="doi">10.1037/0022-3514.67.2.319</pub-id></citation></ref>
<ref id="B13">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Collins</surname> <given-names>A. G. E.</given-names></name> <name><surname>Frank</surname> <given-names>M. J.</given-names></name></person-group> (<year>2012</year>). <article-title>How much of reinforcement learning is working memory, not reinforcement learning? A behaviorbehavioural, computational, and neurogenetic analysis</article-title>. <source>Eur. J. Neurosci.</source> <volume>35</volume>, <fpage>1024</fpage>&#x02013;<lpage>1035</lpage>. <pub-id pub-id-type="doi">10.1111/j.1460-9568.2011.07980.x</pub-id></citation></ref>
<ref id="B14">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Collins</surname> <given-names>A. G. E.</given-names></name> <name><surname>Ciullo</surname> <given-names>B.</given-names></name> <name><surname>Frank</surname> <given-names>M. J.</given-names></name> <name><surname>Badre</surname> <given-names>D.</given-names></name></person-group> (<year>2017</year>). <article-title>Working memory load strengthens reward prediction errors</article-title>. <source>J. Neurosci.</source> <volume>37</volume>, <fpage>4332</fpage>&#x02013;<lpage>4342</lpage>. <pub-id pub-id-type="doi">10.1523/JNEUROSCI.2700-16.2017</pub-id><pub-id pub-id-type="pmid">28320846</pub-id></citation></ref>
<ref id="B15">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Coppin</surname> <given-names>G.</given-names></name> <name><surname>Nolan-Poupart</surname> <given-names>S.</given-names></name> <name><surname>Jones-Gotman</surname> <given-names>M.</given-names></name> <name><surname>Small</surname> <given-names>D. M.</given-names></name></person-group> (<year>2014</year>). <article-title>Working memory and reward association learning impairments in obesity</article-title>. <source>Neuropsychologia</source> <volume>65</volume>, <fpage>146</fpage>&#x02013;<lpage>155</lpage>. <pub-id pub-id-type="doi">10.1016/j.neuropsychologia.2014.10.004</pub-id><pub-id pub-id-type="pmid">25447070</pub-id></citation></ref>
<ref id="B16">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Crone</surname> <given-names>E. A.</given-names></name> <name><surname>Bunge</surname> <given-names>S. A.</given-names></name> <name><surname>de Klerk</surname> <given-names>P.</given-names></name> <name><surname>van der Molen</surname> <given-names>M. W.</given-names></name></person-group> (<year>2005</year>). <article-title>Cardiac concomitants of performance monitoring: context dependence and individual differences</article-title>. <source>Brain Res. Cogn. Brain Res</source>. <volume>23</volume>, <fpage>93</fpage>&#x02013;<lpage>106</lpage>. <pub-id pub-id-type="doi">10.1016/j.cogbrainres.2005.01.009</pub-id><pub-id pub-id-type="pmid">15795137</pub-id></citation></ref>
<ref id="B17">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Crone</surname> <given-names>E. A.</given-names></name> <name><surname>Jennings</surname> <given-names>J. R.</given-names></name> <name><surname>Van der Molen</surname> <given-names>M. W.</given-names></name></person-group> (<year>2004a</year>). <article-title>Developmental change in feedback processing as reflected by phasic heart rate changes</article-title>. <source>Dev. Psychol.</source> <volume>40</volume>, <fpage>1228</fpage>&#x02013;<lpage>1238</lpage>. <pub-id pub-id-type="doi">10.1037/0012-1649.40.6.1228</pub-id><pub-id pub-id-type="pmid">15535769</pub-id></citation></ref>
<ref id="B18">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Crone</surname> <given-names>E. A.</given-names></name> <name><surname>Somsen</surname> <given-names>R. J.</given-names></name> <name><surname>Van Beek</surname> <given-names>B.</given-names></name> <name><surname>Van Der Molen</surname> <given-names>M. W.</given-names></name></person-group> (<year>2004b</year>). <article-title>Heart rate and skin conductance analysis of antecendents and consequences of decision making</article-title>. <source>Psychophysiology</source> <volume>41</volume>, <fpage>531</fpage>&#x02013;<lpage>540</lpage>. <pub-id pub-id-type="doi">10.1111/j.1469-8986.2004.00197.x</pub-id><pub-id pub-id-type="pmid">15189476</pub-id></citation></ref>
<ref id="B19">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Crone</surname> <given-names>E. A.</given-names></name> <name><surname>van der Veen</surname> <given-names>F. M.</given-names></name> <name><surname>van der Molen</surname> <given-names>M. W.</given-names></name> <name><surname>Somsen</surname> <given-names>R. J.</given-names></name> <name><surname>van Beek</surname> <given-names>B.</given-names></name> <name><surname>Jennings</surname> <given-names>J. R.</given-names></name></person-group> (<year>2003</year>). <article-title>Cardiac concomitants of feedback processing</article-title>. <source>Biol Psychol</source>. <volume>64</volume>, <fpage>143</fpage>&#x02013;<lpage>156</lpage>. <pub-id pub-id-type="doi">10.1016/S0301-0511(03)00106-6</pub-id><pub-id pub-id-type="pmid">14602359</pub-id></citation></ref>
<ref id="B20">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>de Ruiter</surname> <given-names>M. B.</given-names></name> <name><surname>Veltman</surname> <given-names>D. J.</given-names></name> <name><surname>Goudriaan</surname> <given-names>A. E.</given-names></name> <name><surname>Oosterlaan</surname> <given-names>J.</given-names></name> <name><surname>Sjoerds</surname> <given-names>Z.</given-names></name> <name><surname>van den Brink</surname> <given-names>W.</given-names></name></person-group> (<year>2009</year>). <article-title>Response perseveration and ventral prefrontal sensitivity to reward and punishment in male problem gamblers and smokers</article-title>. <source>Neuropsychopharmacology</source> <volume>34</volume>, <fpage>1027</fpage>&#x02013;<lpage>1038</lpage>. <pub-id pub-id-type="doi">10.1038/npp.2008.175</pub-id><pub-id pub-id-type="pmid">18830241</pub-id></citation></ref>
<ref id="B21">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>de Weijer</surname> <given-names>B. A.</given-names></name> <name><surname>van de Giessen</surname> <given-names>E.</given-names></name> <name><surname>van Amelsvoort</surname> <given-names>T. A.</given-names></name> <name><surname>Boot</surname> <given-names>E.</given-names></name> <name><surname>Braak</surname> <given-names>B.</given-names></name> <name><surname>Janssen</surname> <given-names>I. M.</given-names></name> <etal/></person-group>. (<year>2011</year>). <article-title>Lower striatal dopamine D2/3 receptor availability in obese compared with non-obese subjects</article-title>. <source>EJNMMI Res</source> <volume>1</volume>:<fpage>37</fpage>. <pub-id pub-id-type="doi">10.1186/2191-219X-1-37</pub-id><pub-id pub-id-type="pmid">22214469</pub-id></citation></ref>
<ref id="B22">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Delgado</surname> <given-names>M. R.</given-names></name> <name><surname>Li</surname> <given-names>J.</given-names></name> <name><surname>Schiller</surname> <given-names>D.</given-names></name> <name><surname>Phelps</surname> <given-names>E. A.</given-names></name></person-group> (<year>2008</year>). <article-title>The role of the striatum in aversive learning and aversive prediction errors</article-title>. <source>Philos. Trans. R. Soc. B Biol. Sci.</source> <volume>363</volume>, <fpage>3787</fpage>&#x02013;<lpage>3800</lpage>. <pub-id pub-id-type="doi">10.1098/rstb.2008.0161</pub-id><pub-id pub-id-type="pmid">18829426</pub-id></citation></ref>
<ref id="B23">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Delgado</surname> <given-names>M. R.</given-names></name> <name><surname>Nystrom</surname> <given-names>L. E.</given-names></name> <name><surname>Fissell</surname> <given-names>C.</given-names></name> <name><surname>Noll</surname> <given-names>D. C.</given-names></name> <name><surname>Fiez</surname> <given-names>J. A.</given-names></name></person-group> (<year>2000</year>). <article-title>Tracking the hemodynamic responses to reward and punishment in the striatum</article-title>. <source>J. Neurophysiol</source>. <volume>84</volume>, <fpage>3072</fpage>&#x02013;<lpage>3077</lpage>. <pub-id pub-id-type="pmid">11110834</pub-id></citation></ref>
<ref id="B24">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Eppinger</surname> <given-names>B.</given-names></name> <name><surname>Schuck</surname> <given-names>N. W.</given-names></name> <name><surname>Nystrom</surname> <given-names>L. E.</given-names></name> <name><surname>Cohen</surname> <given-names>J. D.</given-names></name></person-group> (<year>2013</year>). <article-title>Reduced striatal responses to reward prediction errors in older compared with younger adults</article-title>. <source>J. Neurosci.</source> <volume>33</volume>, <fpage>9905</fpage>&#x02013;<lpage>9912</lpage>. <pub-id pub-id-type="doi">10.1523/JNEUROSCI.2942-12.2013</pub-id><pub-id pub-id-type="pmid">23761885</pub-id></citation></ref>
<ref id="B25">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Etkin</surname> <given-names>A.</given-names></name> <name><surname>Wager</surname> <given-names>T. D.</given-names></name></person-group> (<year>2007</year>). <article-title>Functional neuroimaging of anxiety: a meta-analysis of emotional processing in PTSD, social anxiety disorder, and specific phobia</article-title>. <source>Am. J. Psychiatry</source> <volume>164</volume>, <fpage>1476</fpage>&#x02013;<lpage>1488</lpage>. <pub-id pub-id-type="doi">10.1176/appi.ajp.2007.07030504</pub-id><pub-id pub-id-type="pmid">17898336</pub-id></citation></ref>
<ref id="B26">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Evans</surname> <given-names>K. L.</given-names></name> <name><surname>Hampson</surname> <given-names>E.</given-names></name></person-group> (<year>2015</year>). <article-title>Sex differences on prefrontally-dependent cognitive tasks</article-title>. <source>Brain Cogn.</source> <volume>93</volume>, <fpage>42</fpage>&#x02013;<lpage>53</lpage>. <pub-id pub-id-type="doi">10.1016/j.bandc.2014.11.006</pub-id><pub-id pub-id-type="pmid">25528435</pub-id></citation></ref>
<ref id="B27">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Figley</surname> <given-names>C. R.</given-names></name> <name><surname>Asem</surname> <given-names>J. S.</given-names></name> <name><surname>Levenbaum</surname> <given-names>E. L.</given-names></name> <name><surname>Courtney</surname> <given-names>S. M.</given-names></name></person-group> (<year>2016</year>). <article-title>Effects of body mass index and body fat percent on default mode, executive control, and salience network structure and function</article-title>. <source>Front. Neurosci</source>. <volume>10</volume>:<fpage>234</fpage>. <pub-id pub-id-type="doi">10.3389/fnins.2016.00234</pub-id><pub-id pub-id-type="pmid">27378831</pub-id></citation></ref>
<ref id="B28">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Frank</surname> <given-names>M. J.</given-names></name> <name><surname>Seeberger</surname> <given-names>L. C.</given-names></name> <name><surname>O&#x00027;reilly</surname> <given-names>R. C.</given-names></name></person-group> (<year>2004</year>). <article-title>By carrot or by stick: cognitive reinforcement learning in parkinsonism</article-title>. <source>Science</source> <volume>306</volume>, <fpage>1940</fpage>&#x02013;<lpage>1943</lpage>. <pub-id pub-id-type="doi">10.1126/science.1102941</pub-id><pub-id pub-id-type="pmid">15528409</pub-id></citation></ref>
<ref id="B29">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Franken</surname> <given-names>I. H. A.</given-names></name> <name><surname>Muris</surname> <given-names>P.</given-names></name></person-group> (<year>2005</year>). <article-title>Individual differences in decision-making</article-title>. <source>Pers. Individ. Dif.</source> <volume>39</volume>, <fpage>991</fpage>&#x02013;<lpage>998</lpage>. <pub-id pub-id-type="doi">10.1016/j.paid.2005.04.004</pub-id></citation></ref>
<ref id="B30">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Franken</surname> <given-names>I. H. A.</given-names></name> <name><surname>van Strien</surname> <given-names>J. W.</given-names></name> <name><surname>Nijs</surname> <given-names>I.</given-names></name> <name><surname>Muris</surname> <given-names>P.</given-names></name></person-group> (<year>2008</year>). <article-title>Impulsivity is associated with behavioural decision-making deficits</article-title>. <source>Psychiatry Res.</source> <volume>158</volume>, <fpage>155</fpage>&#x02013;<lpage>163</lpage>. <pub-id pub-id-type="doi">10.1016/j.psychres.2007.06.002</pub-id><pub-id pub-id-type="pmid">18215765</pub-id></citation></ref>
<ref id="B31">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Garc&#x000ED;a-Garc&#x000ED;a</surname> <given-names>I.</given-names></name> <name><surname>Horstmann</surname> <given-names>A.</given-names></name> <name><surname>Jurado</surname> <given-names>M. A.</given-names></name> <name><surname>Garolera</surname> <given-names>M.</given-names></name> <name><surname>Chaudhry</surname> <given-names>S. J.</given-names></name> <name><surname>Margulies</surname> <given-names>D. S.</given-names></name> <etal/></person-group>. (<year>2014</year>). <article-title>Reward processing in obesity, substance addiction and non-substance addiction</article-title>. <source>Obesity Rev.</source> <volume>15</volume>, <fpage>853</fpage>&#x02013;<lpage>869</lpage>. <pub-id pub-id-type="doi">10.1111/obr.12221</pub-id><pub-id pub-id-type="pmid">25263466</pub-id></citation></ref>
<ref id="B32">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Garc&#x000ED;a-Garc&#x000ED;a</surname> <given-names>I.</given-names></name> <name><surname>Jurado</surname> <given-names>M. &#x000C1;.</given-names></name> <name><surname>Garolera</surname> <given-names>M.</given-names></name> <name><surname>Marqu&#x000E9;s-Iturria</surname> <given-names>I.</given-names></name> <name><surname>Horstmann</surname> <given-names>A.</given-names></name> <name><surname>Segura</surname> <given-names>B.</given-names></name> <etal/></person-group>. (<year>2015</year>). <article-title>Functional network centrality in obesity: a resting-state and task fMRI study</article-title>. <source>Psychiatry Res.</source> <volume>233</volume>, <fpage>331</fpage>&#x02013;<lpage>338</lpage>. <pub-id pub-id-type="doi">10.1016/j.pscychresns.2015.05.017</pub-id><pub-id pub-id-type="pmid">26145769</pub-id></citation></ref>
<ref id="B33">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Garfinkel</surname> <given-names>S. N.</given-names></name> <name><surname>Barrett</surname> <given-names>A. B.</given-names></name> <name><surname>Minati</surname> <given-names>L.</given-names></name> <name><surname>Dolan</surname> <given-names>R. J.</given-names></name> <name><surname>Seth</surname> <given-names>A. K.</given-names></name> <name><surname>Critchley</surname> <given-names>H. D.</given-names></name></person-group> (<year>2013</year>). <article-title>What the heart forgets: cardiac timing influences memory for words and is modulated by metacognition and interoceptive sensitivity</article-title>. <source>Psychophysiology</source> <volume>50</volume>, <fpage>505</fpage>&#x02013;<lpage>512</lpage>. <pub-id pub-id-type="doi">10.1111/psyp.12039</pub-id><pub-id pub-id-type="pmid">23521494</pub-id></citation></ref>
<ref id="B34">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Garfinkel</surname> <given-names>S. N.</given-names></name> <name><surname>Minati</surname> <given-names>L.</given-names></name> <name><surname>Gray</surname> <given-names>M. A.</given-names></name> <name><surname>Seth</surname> <given-names>A. K.</given-names></name> <name><surname>Dolan</surname> <given-names>R. J.</given-names></name> <name><surname>Critchley</surname> <given-names>H. D.</given-names></name></person-group> (<year>2014</year>). <article-title>Fear from the heart: sensitivity to fear stimuli depends on individual heartbeats</article-title>. <source>J. Neurosci.</source> <volume>34</volume>, <fpage>6573</fpage>&#x02013;<lpage>6582</lpage>. <pub-id pub-id-type="doi">10.1523/JNEUROSCI.3507-13.2014</pub-id><pub-id pub-id-type="pmid">24806682</pub-id></citation></ref>
<ref id="B35">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gl&#x000E4;scher</surname> <given-names>J. P.</given-names></name> <name><surname>O&#x00027;Doherty</surname> <given-names>J. P.</given-names></name></person-group> (<year>2010</year>). <article-title>Model-based approaches to neuroimaging: combining reinforcement learning theory with fMRI data</article-title>. <source>Wiley Interdiscip. Rev. Cogn. Sci</source>. <volume>1</volume>, <fpage>501</fpage>&#x02013;<lpage>510</lpage>. <pub-id pub-id-type="doi">10.1002/wcs.57</pub-id><pub-id pub-id-type="pmid">26271497</pub-id></citation></ref>
<ref id="B36">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Glimcher</surname> <given-names>P. W.</given-names></name></person-group> (<year>2011</year>). <article-title>Understanding dopamine and reinforcement learning: the dopamine reward prediction error hypothesis</article-title>. <source>Proc. Natl. Acad. Sci. U. S. A</source>. <volume>108</volume>, <fpage>15647</fpage>&#x02013;<lpage>15654</lpage>. <pub-id pub-id-type="doi">10.1073/pnas.1014269108</pub-id><pub-id pub-id-type="pmid">21389268</pub-id></citation></ref>
<ref id="B37">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gradin</surname> <given-names>V. B.</given-names></name> <name><surname>Kumar</surname> <given-names>P.</given-names></name> <name><surname>Waiter</surname> <given-names>G.</given-names></name> <name><surname>Ahearn</surname> <given-names>T.</given-names></name> <name><surname>Stickle</surname> <given-names>C.</given-names></name> <name><surname>Milders</surname> <given-names>M.</given-names></name> <etal/></person-group>. (<year>2011</year>). <article-title>Expected value and prediction error abnormalities in depression and schizophrenia</article-title>. <source>Brain</source> <volume>134</volume>, <fpage>1751</fpage>&#x02013;<lpage>1764</lpage>. <pub-id pub-id-type="doi">10.1093/brain/awr059</pub-id><pub-id pub-id-type="pmid">21482548</pub-id></citation></ref>
<ref id="B38">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Gray</surname> <given-names>M. A.</given-names></name> <name><surname>Beacher</surname> <given-names>F. D.</given-names></name> <name><surname>Minati</surname> <given-names>L.</given-names></name> <name><surname>Nagai</surname> <given-names>Y.</given-names></name> <name><surname>Kemp</surname> <given-names>A. H.</given-names></name> <name><surname>Harrison</surname> <given-names>N.</given-names></name> <etal/></person-group>. (<year>2012</year>). <article-title>Emotional appraisal is influenced by cardiac afferent information</article-title>. <source>Emotion</source> <volume>12</volume>, <fpage>180</fpage>&#x02013;<lpage>191</lpage>. <pub-id pub-id-type="doi">10.1037/a0025083</pub-id><pub-id pub-id-type="pmid">21988743</pub-id></citation></ref>
<ref id="B39">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Groen</surname> <given-names>Y.</given-names></name> <name><surname>Wijers</surname> <given-names>A. A.</given-names></name> <name><surname>Mulder</surname> <given-names>L. J. M.</given-names></name> <name><surname>Minderaa</surname> <given-names>R. B.</given-names></name> <name><surname>Althaus</surname> <given-names>M.</given-names></name></person-group> (<year>2007</year>). <article-title>Physiological correlates of learning by performance feedback in children: a study of EEG event-related potentials and evoked heart rate</article-title>. <source>Biol. Psychol.</source> <volume>76</volume>, <fpage>174</fpage>&#x02013;<lpage>187</lpage>. <pub-id pub-id-type="doi">10.1016/j.biopsycho.2007.07.006</pub-id><pub-id pub-id-type="pmid">17888560</pub-id></citation></ref>
<ref id="B40">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Guitart-Masip</surname> <given-names>M.</given-names></name> <name><surname>Economides</surname> <given-names>M.</given-names></name> <name><surname>Huys</surname> <given-names>Q. J.</given-names></name> <name><surname>Frank</surname> <given-names>M. J.</given-names></name> <name><surname>Chowdhury</surname> <given-names>R.</given-names></name> <name><surname>Duzel</surname> <given-names>E.</given-names></name> <etal/></person-group>. (<year>2014</year>). <article-title>Differential, but not opponent, effects of L -DOPA and citalopram on action learning with reward and punishment</article-title>. <source>Psychopharmacology</source> <volume>231</volume>, <fpage>955</fpage>&#x02013;<lpage>966</lpage>. <pub-id pub-id-type="doi">10.1007/s00213-013-3313-4</pub-id></citation></ref>
<ref id="B41">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>H&#x000E4;mmerer</surname> <given-names>D.</given-names></name> <name><surname>Eppinger</surname> <given-names>B.</given-names></name></person-group> (<year>2012</year>). <article-title>Dopaminergic and prefrontal contributions to reward-based learning and outcome monitoring during child development and aging</article-title>. <source>Dev. Psychol</source>. <volume>48</volume>, <fpage>862</fpage>&#x02013;<lpage>874</lpage>. <pub-id pub-id-type="doi">10.1037/a0027342</pub-id><pub-id pub-id-type="pmid">22390655</pub-id></citation></ref>
<ref id="B42">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>H&#x000E4;rting</surname> <given-names>C.</given-names></name> <name><surname>Markowitsch</surname> <given-names>H.-J.</given-names></name> <name><surname>Neufeld</surname> <given-names>H.</given-names></name> <name><surname>Calabrese</surname> <given-names>P.</given-names></name> <name><surname>Deisinger</surname> <given-names>K.</given-names></name> <name><surname>Kessler</surname> <given-names>J.</given-names></name></person-group> (<year>2000</year>). <source>Wechsler Memory Scale - Revised Edition, German Edition</source>. <publisher-loc>Bern</publisher-loc>: <publisher-name>Huber</publisher-name>.</citation></ref>
<ref id="B43">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hogenkamp</surname> <given-names>P. S.</given-names></name> <name><surname>Zhou</surname> <given-names>W. L.</given-names></name> <name><surname>Dahlberg</surname> <given-names>L. S.</given-names></name> <name><surname>Stark</surname> <given-names>J.</given-names></name> <name><surname>Larsen</surname> <given-names>A. L.</given-names></name> <name><surname>Olivo</surname> <given-names>G.</given-names></name> <etal/></person-group>. (<year>2016</year>). <article-title>Higher resting-state activity in reward-related brain circuits in obese versus normal-weight females independent of food intake</article-title>. <source>Int. J. Obes.</source> <volume>40</volume>, <fpage>1687</fpage>&#x02013;<lpage>1692</lpage>. <pub-id pub-id-type="doi">10.1038/ijo.2016.105</pub-id><pub-id pub-id-type="pmid">27349694</pub-id></citation></ref>
<ref id="B44">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Horstmann</surname> <given-names>A.</given-names></name> <name><surname>Busse</surname> <given-names>F. P.</given-names></name> <name><surname>Mathar</surname> <given-names>D.</given-names></name> <name><surname>M&#x000FC;ller</surname> <given-names>K.</given-names></name> <name><surname>Lepsien</surname> <given-names>J.</given-names></name> <name><surname>Schl&#x000F6;gl</surname> <given-names>H.</given-names></name> <etal/></person-group>. (<year>2011</year>). <article-title>Obesity-related differences between women and men in brain structure and goal-directed behavior</article-title>. <source>Front. Hum. Neurosci.</source> <volume>5</volume>:<fpage>58</fpage>. <pub-id pub-id-type="doi">10.3389/fnhum.2011.00058</pub-id><pub-id pub-id-type="pmid">21713067</pub-id></citation></ref>
<ref id="B45">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Horstmann</surname> <given-names>A.</given-names></name> <name><surname>Dietrich</surname> <given-names>A.</given-names></name> <name><surname>Mathar</surname> <given-names>D.</given-names></name> <name><surname>P&#x000F6;ssel</surname> <given-names>M.</given-names></name> <name><surname>Villringer</surname> <given-names>A.</given-names></name> <name><surname>Neumann</surname> <given-names>J.</given-names></name></person-group> (<year>2015a</year>). <article-title>Slave to habit? Obesity is associated with decreased behavioural sensitivity to reward devaluation</article-title>. <source>Appetite</source> <volume>87</volume>, <fpage>175</fpage>&#x02013;<lpage>183</lpage>. <pub-id pub-id-type="doi">10.1016/j.appet.2014.12.212</pub-id><pub-id pub-id-type="pmid">25543077</pub-id></citation></ref>
<ref id="B46">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Horstmann</surname> <given-names>A.</given-names></name> <name><surname>Fenske</surname> <given-names>W. K.</given-names></name> <name><surname>Hankir</surname> <given-names>M. K.</given-names></name></person-group> (<year>2015b</year>). <article-title>Argument for a non-linear relationship between severity of human obesity and dopaminergic tone</article-title>. <source>Obes Rev</source>. <volume>16</volume>, <fpage>821</fpage>&#x02013;<lpage>830</lpage>. <pub-id pub-id-type="doi">10.1111/obr.12303</pub-id><pub-id pub-id-type="pmid">26098597</pub-id></citation></ref>
<ref id="B47">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Hottenrott</surname> <given-names>K.</given-names></name> <name><surname>Hoos</surname> <given-names>O.</given-names></name> <name><surname>Esperer</surname> <given-names>H. D.</given-names></name></person-group> (<year>2006</year>). <article-title>Heart rate variability and physical exercise. Current status</article-title>. <source>Herz</source>. <volume>31</volume>:<fpage>544</fpage>. <pub-id pub-id-type="doi">10.1007/s00059-006-2855-1</pub-id><pub-id pub-id-type="pmid">17036185</pub-id></citation></ref>
<ref id="B48">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Jocham</surname> <given-names>G.</given-names></name> <name><surname>Klein</surname> <given-names>T. A.</given-names></name> <name><surname>Ullsperger</surname> <given-names>M.</given-names></name></person-group> (<year>2014</year>). <article-title>Differential modulation of reinforcement learning by D2 Dopamine and NMDA glutamate receptor antagonism</article-title>. <source>J. Neurosci.</source> <volume>34</volume>, <fpage>13151</fpage>&#x02013;<lpage>13162</lpage>. <pub-id pub-id-type="doi">10.1523/JNEUROSCI.0757-14.2014</pub-id><pub-id pub-id-type="pmid">25253860</pub-id></citation></ref>
<ref id="B49">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Karason</surname> <given-names>K.</given-names></name> <name><surname>M&#x000F8;lgaard</surname> <given-names>H.</given-names></name> <name><surname>Wikstrand</surname> <given-names>J.</given-names></name> <name><surname>Sj&#x000F6;str&#x000F6;m</surname> <given-names>L.</given-names></name></person-group> (<year>1999</year>). <article-title>Heart rate variability in obesity and the effect of weight loss</article-title>. <source>Am. J. Cardiol</source>. <volume>83</volume>, <fpage>1242</fpage>&#x02013;<lpage>1247</lpage>. <pub-id pub-id-type="doi">10.1016/S0002-9149(99)00066-1</pub-id><pub-id pub-id-type="pmid">10215292</pub-id></citation></ref>
<ref id="B50">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kim</surname> <given-names>D.-I.</given-names></name> <name><surname>Yang</surname> <given-names>H. I.</given-names></name> <name><surname>Park</surname> <given-names>J.-H.</given-names></name> <name><surname>Lee</surname> <given-names>M. K.</given-names></name> <name><surname>Kang</surname> <given-names>D.-W.</given-names></name> <name><surname>Chae</surname> <given-names>J. S.</given-names></name> <etal/></person-group>. (<year>2016</year>). <article-title>The association between resting heart rate and type 2 diabetes and hypertension in Korean adults</article-title>. <source>Heart</source> <volume>102</volume>, <fpage>1757</fpage>&#x02013;<lpage>1762</lpage>. <pub-id pub-id-type="doi">10.1136/heartjnl-2015-309119</pub-id><pub-id pub-id-type="pmid">27312000</pub-id></citation></ref>
<ref id="B51">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kim</surname> <given-names>H.</given-names></name> <name><surname>Shimojo</surname> <given-names>S.</given-names></name> <name><surname>O&#x00027;Doherty</surname> <given-names>J. P.</given-names></name></person-group> (<year>2006</year>). <article-title>Is Avoiding an aversive outcome rewarding? Neural substrates of avoidance learning in the human brain</article-title>. <source>PLoS Biol</source>. <volume>4</volume>:<fpage>e233</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pbio.0040233</pub-id><pub-id pub-id-type="pmid">16802856</pub-id></citation></ref>
<ref id="B52">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Klein</surname> <given-names>T. A.</given-names></name> <name><surname>Neumann</surname> <given-names>J.</given-names></name> <name><surname>Reuter</surname> <given-names>M.</given-names></name> <name><surname>Hennig</surname> <given-names>J.</given-names></name> <name><surname>Von Cramon</surname> <given-names>D. Y.</given-names></name> <name><surname>Ullsperger</surname> <given-names>M.</given-names></name></person-group> (<year>2007</year>). <article-title>Genetically determined differences in learning from errors</article-title>. <source>Science</source> <volume>318</volume>, <fpage>1642</fpage>&#x02013;<lpage>1645</lpage>. <pub-id pub-id-type="doi">10.1126/science.1145044</pub-id><pub-id pub-id-type="pmid">18063800</pub-id></citation></ref>
<ref id="B53">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Koenig</surname> <given-names>J.</given-names></name> <name><surname>Thayer</surname> <given-names>J. F.</given-names></name></person-group> (<year>2016</year>). <article-title>Sex differences in healthy human heart rate variability: a meta-analysis</article-title>. <source>Neurosci. Biobehav. Rev</source>. <volume>64</volume>, <fpage>288</fpage>&#x02013;<lpage>310</lpage>. <pub-id pub-id-type="doi">10.1016/j.neubiorev.2016.03.007</pub-id><pub-id pub-id-type="pmid">26964804</pub-id></citation></ref>
<ref id="B54">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kube</surname> <given-names>J.</given-names></name> <name><surname>Schrimpf</surname> <given-names>A.</given-names></name> <name><surname>Garc&#x000ED;a-Garc&#x000ED;a</surname> <given-names>I.</given-names></name> <name><surname>Villringer</surname> <given-names>A.</given-names></name> <name><surname>Neumann</surname> <given-names>J.</given-names></name> <name><surname>Horstmann</surname> <given-names>A.</given-names></name></person-group> (<year>2016</year>). <article-title>Differential heart rate responses to social and monetary reinforcement in women with obesity</article-title>. <source>Psychophysiology</source> <volume>53</volume>, <fpage>868</fpage>&#x02013;<lpage>879</lpage>. <pub-id pub-id-type="doi">10.1111/psyp.12624</pub-id><pub-id pub-id-type="pmid">26871590</pub-id></citation></ref>
<ref id="B55">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Kullmann</surname> <given-names>S.</given-names></name> <name><surname>Heni</surname> <given-names>M.</given-names></name> <name><surname>Veit</surname> <given-names>R.</given-names></name> <name><surname>Ketterer</surname> <given-names>C.</given-names></name> <name><surname>Schick</surname> <given-names>F.</given-names></name> <name><surname>H&#x000E4;ring</surname> <given-names>H. U.</given-names></name> <etal/></person-group>. (<year>2011</year>). <article-title>The obese brain: association of body mass index and insulin sensitivity with resting state network functional connectivity</article-title>. <source>Hum. Brain Mapp.</source> <volume>33</volume>, <fpage>1052</fpage>&#x02013;<lpage>1061</lpage>. <pub-id pub-id-type="doi">10.1002/hbm.21268</pub-id><pub-id pub-id-type="pmid">21520345</pub-id></citation></ref>
<ref id="B56">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Laborde</surname> <given-names>S.</given-names></name> <name><surname>Mosley</surname> <given-names>E.</given-names></name> <name><surname>Thayer</surname> <given-names>J. F.</given-names></name></person-group> (<year>2017</year>). <article-title>Heart rate variability and cardiac vagal tone in psychophysiological research &#x02013; recommendations for experiment planning, data analysis, and data reporting</article-title>. <source>Front. Psychol.</source> <volume>8</volume>:<fpage>213</fpage>. <pub-id pub-id-type="doi">10.3389/fpsyg.2017.00213</pub-id><pub-id pub-id-type="pmid">28265249</pub-id></citation></ref>
<ref id="B57">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Lee</surname> <given-names>S. W.</given-names></name> <name><surname>Shimojo</surname> <given-names>S.</given-names></name> <name><surname>O&#x00027;Doherty</surname> <given-names>J. P.</given-names></name></person-group> (<year>2014</year>). <article-title>Neural computations underlying arbitration between model-based and model-free learning</article-title>. <source>Neuron</source> <volume>81</volume>, <fpage>687</fpage>&#x02013;<lpage>699</lpage>. <pub-id pub-id-type="doi">10.1016/j.neuron.2013.11.028</pub-id>. <pub-id pub-id-type="pmid">24507199</pub-id></citation></ref>
<ref id="B58">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Levy</surname> <given-names>D. J.</given-names></name> <name><surname>Thavikulwat</surname> <given-names>A. C.</given-names></name> <name><surname>Glimcher</surname> <given-names>P. W.</given-names></name></person-group> (<year>2013</year>). <article-title>State dependent valuation: the effect of deprivation on risk preferences</article-title>. <source>PLoS ONE</source> <volume>8</volume>:<fpage>e53978</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0053978</pub-id><pub-id pub-id-type="pmid">23358126</pub-id></citation></ref>
<ref id="B59">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Maia</surname> <given-names>T. V.</given-names></name> <name><surname>Frank</surname> <given-names>M. J.</given-names></name></person-group> (<year>2011</year>). <article-title>From reinforcement learning models to psychiatric and neurological disorders</article-title>. <source>Nat. Neurosci</source>. <volume>14</volume>, <fpage>154</fpage>&#x02013;<lpage>162</lpage>. <pub-id pub-id-type="doi">10.1038/nn.2723</pub-id><pub-id pub-id-type="pmid">21270784</pub-id></citation></ref>
<ref id="B60">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Mancia</surname> <given-names>G.</given-names></name> <name><surname>Fagard</surname> <given-names>R.</given-names></name> <name><surname>Narkiewicz</surname> <given-names>K.</given-names></name> <name><surname>Red&#x000F3;n</surname> <given-names>J.</given-names></name> <name><surname>Zanchetti</surname> <given-names>A.</given-names></name> <name><surname>B&#x000F6;hm</surname> <given-names>M.</given-names></name> <etal/></person-group>. (<year>2013</year>). <article-title>2013 ESH/ESC Guidelines for the management of arterial hypertension: the Task Force for the management of arterial hypertension of the European Society of Hypertension (ESH) and of the European Society of Cardiology (ESC)</article-title>. <source>J. Hypertens</source>. <volume>7</volume>, <fpage>1281</fpage>&#x02013;<lpage>357</lpage>. <pub-id pub-id-type="doi">10.1097/01.hjh.0000431740.32696.cc</pub-id></citation></ref>
<ref id="B61">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Martin</surname> <given-names>L. E.</given-names></name> <name><surname>Cox</surname> <given-names>L. S.</given-names></name> <name><surname>Brooks</surname> <given-names>W. M.</given-names></name> <name><surname>Savage</surname> <given-names>C. R.</given-names></name></person-group> (<year>2014</year>). <article-title>Winning and losing: differences in reward and punishment sensitivity between smokers and nonsmokers</article-title>. <source>Brain Behav.</source> <volume>4</volume>, <fpage>915</fpage>&#x02013;<lpage>924</lpage>. <pub-id pub-id-type="doi">10.1002/brb3.285</pub-id><pub-id pub-id-type="pmid">25365800</pub-id></citation></ref>
<ref id="B62">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Mathar</surname> <given-names>D.</given-names></name> <name><surname>Horstmann</surname> <given-names>A.</given-names></name> <name><surname>Pleger</surname> <given-names>B.</given-names></name> <name><surname>Villringer</surname> <given-names>A.</given-names></name> <name><surname>Neumann</surname> <given-names>J.</given-names></name></person-group> (<year>2016</year>). <article-title>Is it worth the effort? Novel insights into obesity-associated alterations in cost-benefit decision-making</article-title>. <source>Front. Behav. Neurosci</source> <volume>9</volume>:<fpage>360</fpage>. <pub-id pub-id-type="doi">10.3389/fnbeh.2015.00360</pub-id><pub-id pub-id-type="pmid">26793079</pub-id></citation></ref>
<ref id="B63">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Mathar</surname> <given-names>D.</given-names></name> <name><surname>Neumann</surname> <given-names>J.</given-names></name> <name><surname>Villringer</surname> <given-names>A.</given-names></name> <name><surname>Horstmann</surname> <given-names>A.</given-names></name></person-group> (<year>2017a</year>). <article-title>Failing to learn from negative prediction errors: obesity is associated with alterations in a fundamental neural learning mechanism</article-title>. <source>Cortex</source> <volume>95</volume>, <fpage>222</fpage>&#x02013;<lpage>237</lpage>. <pub-id pub-id-type="doi">10.1016/j.cortex.2017.08.022</pub-id><pub-id pub-id-type="pmid">28917135</pub-id></citation></ref>
<ref id="B64">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Mathar</surname> <given-names>D.</given-names></name> <name><surname>Wilkinson</surname> <given-names>L.</given-names></name> <name><surname>Holl</surname> <given-names>A.</given-names></name> <name><surname>Neumann</surname> <given-names>J.</given-names></name> <name><surname>Deserno</surname> <given-names>L.</given-names></name> <name><surname>Villringer</surname> <given-names>A.</given-names></name> <etal/></person-group>. (<year>2017b</year>). <article-title>The role of dopamine in positive and negative prediction error utilization during incidental learning - insights from positron emission tomography, Parkinson&#x00027;s disease and Huntington&#x00027;s disease</article-title>. <source>Cortex</source> <volume>90</volume>, <fpage>149</fpage>&#x02013;<lpage>162</lpage>. <pub-id pub-id-type="doi">10.1016/j.cortex.2016.09.004</pub-id><pub-id pub-id-type="pmid">27751503</pub-id></citation></ref>
<ref id="B65">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Mattfeld</surname> <given-names>A. T.</given-names></name> <name><surname>Gluck</surname> <given-names>M. A.</given-names></name> <name><surname>Stark</surname> <given-names>C. E.</given-names></name></person-group> (<year>2011</year>). <article-title>Functional specialization within the striatum along both the dorsal/ventral and anterior/posterior axes during associative learning via reward and punishment</article-title>. <source>Learn Mem</source>. <volume>18</volume>, <fpage>703</fpage>&#x02013;<lpage>711</lpage>. <pub-id pub-id-type="doi">10.1101/lm.022889.111</pub-id><pub-id pub-id-type="pmid">22021252</pub-id></citation></ref>
<ref id="B66">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Mies</surname> <given-names>G. W.</given-names></name> <name><surname>Van der Veen</surname> <given-names>F. M.</given-names></name> <name><surname>Tulen</surname> <given-names>J.</given-names></name> <name><surname>Hengeveld</surname> <given-names>M.</given-names></name> <name><surname>Van der Molen</surname> <given-names>M. W.</given-names></name></person-group> (<year>2011</year>). <article-title>Cardiac and electrophysiological repsonses to valid and invalid feedback in a time-estimation task</article-title>. <source>J. Psychophysiol.</source> <volume>25</volume>, <fpage>131</fpage>&#x02013;<lpage>142</lpage>. <pub-id pub-id-type="doi">10.1027/0269-8803/a000049</pub-id></citation></ref>
<ref id="B67">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Mueller</surname> <given-names>A.</given-names></name> <name><surname>Smits</surname> <given-names>D.</given-names></name> <name><surname>Claes</surname> <given-names>L.</given-names></name> <name><surname>de Zwaan</surname> <given-names>M.</given-names></name></person-group> (<year>2013</year>). <article-title>Factor structure of the German version of the BIS/BAS Scales in a population-based sample</article-title>. <source>Fortschr Neurol Psychiatr</source>. <volume>81</volume>, <fpage>75</fpage>&#x02013;<lpage>80</lpage>. <pub-id pub-id-type="doi">10.1055/s-0032-1330482</pub-id></citation></ref>
<ref id="B68">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>M&#x000FC;hlberg</surname> <given-names>C.</given-names></name> <name><surname>Mathar</surname> <given-names>D.</given-names></name> <name><surname>Villringer</surname> <given-names>A.</given-names></name> <name><surname>Horstmann</surname> <given-names>A.</given-names></name> <name><surname>Neumann</surname> <given-names>J.</given-names></name></person-group> (<year>2016</year>). <article-title>Stopping at the sight of food - how gender and obesity impact on response inhibition</article-title>. <source>Appetite</source> <volume>107</volume>, <fpage>663</fpage>&#x02013;<lpage>676</lpage>. <pub-id pub-id-type="doi">10.1016/j.appet.2016.08.121</pub-id><pub-id pub-id-type="pmid">27592420</pub-id></citation></ref>
<ref id="B69">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Opel</surname> <given-names>N.</given-names></name> <name><surname>Redlich</surname> <given-names>R.</given-names></name> <name><surname>Grotegerd</surname> <given-names>D.</given-names></name> <name><surname>Dohm</surname> <given-names>K.</given-names></name> <name><surname>Haupenthal</surname> <given-names>C.</given-names></name> <name><surname>Heindel</surname> <given-names>W.</given-names></name> <etal/></person-group>. (<year>2015</year>). <article-title>Enhanced neural responsiveness to reward associated with obesity in the absence of food-related stimuli</article-title>. <source>Hum. Brain Mapp.</source> <volume>36</volume>, <fpage>2330</fpage>&#x02013;<lpage>2337</lpage>. <pub-id pub-id-type="doi">10.1002/hbm.22773</pub-id><pub-id pub-id-type="pmid">25704752</pub-id></citation></ref>
<ref id="B70">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Overman</surname> <given-names>W. H.</given-names></name></person-group> (<year>2004</year>). <article-title>Sex differences in early childhood, adolescence, and adulthood on cognitive tasks that rely on orbital prefrontal cortex</article-title>. <source>Brain Cogn.</source> <volume>55</volume>, <fpage>134</fpage>&#x02013;<lpage>147</lpage>. <pub-id pub-id-type="doi">10.1016/S0278-2626(03)00279-3</pub-id><pub-id pub-id-type="pmid">15134848</pub-id></citation></ref>
<ref id="B71">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Park</surname> <given-names>S. Q.</given-names></name> <name><surname>Kahnt</surname> <given-names>T.</given-names></name> <name><surname>Beck</surname> <given-names>A.</given-names></name> <name><surname>Cohen</surname> <given-names>M. X.</given-names></name> <name><surname>Dolan</surname> <given-names>R. J.</given-names></name> <name><surname>Wrase</surname> <given-names>J.</given-names></name> <etal/></person-group>. (<year>2010</year>). <article-title>Prefrontal cortex fails to learn from reward prediction errors in alcohol dependence</article-title>. <source>J. Neurosci.</source> <volume>30</volume>, <fpage>7749</fpage>&#x02013;<lpage>7753</lpage>. <pub-id pub-id-type="doi">10.1523/JNEUROSCI.5587-09.2010</pub-id><pub-id pub-id-type="pmid">20519550</pub-id></citation></ref>
<ref id="B72">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Pursey</surname> <given-names>K. M.</given-names></name> <name><surname>Stanwell</surname> <given-names>P.</given-names></name> <name><surname>Callister</surname> <given-names>R. J.</given-names></name> <name><surname>Brain</surname> <given-names>K.</given-names></name> <name><surname>Collins</surname> <given-names>C. E.</given-names></name> <name><surname>Burrows</surname> <given-names>T. L.</given-names></name></person-group> (<year>2014</year>). <article-title>Neural responses to visual food cues according to weight status: a systematic review of functional magnetic resonance imaging studies</article-title>. <source>Front. Nutr.</source> <volume>1</volume>:<fpage>7</fpage>. <pub-id pub-id-type="doi">10.3389/fnut.2014.00007</pub-id><pub-id pub-id-type="pmid">25988110</pub-id></citation></ref>
<ref id="B73">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Quintana</surname> <given-names>D. S.</given-names></name> <name><surname>Heathers</surname> <given-names>J. A. J.</given-names></name></person-group> (<year>2014</year>). <article-title>Considerations in the assessment of heart rate variability in biobehavioral research</article-title>. <source>Front. Psychol.</source> <volume>5</volume>:<fpage>805</fpage>. <pub-id pub-id-type="doi">10.3389/fpsyg.2014.00805</pub-id><pub-id pub-id-type="pmid">25101047</pub-id></citation></ref>
<ref id="B74">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Ramaekers</surname> <given-names>D.</given-names></name> <name><surname>Ector</surname> <given-names>H.</given-names></name> <name><surname>Aubert</surname> <given-names>A. E.</given-names></name> <name><surname>Rubens</surname> <given-names>A.</given-names></name> <name><surname>Van de Werf</surname> <given-names>F.</given-names></name></person-group> (<year>1998</year>). <article-title>Heart rate variability and heart rate in healthy volunteers. Is the female autonomic nervous system cardioprotective?</article-title> <source>Eur. Heart J</source>. <volume>19</volume>, <fpage>1334</fpage>&#x02013;<lpage>1341</lpage>. <pub-id pub-id-type="doi">10.1053/euhj.1998.1084</pub-id><pub-id pub-id-type="pmid">9792258</pub-id></citation></ref>
<ref id="B75">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Rennie</surname> <given-names>K. L.</given-names></name> <name><surname>Hemingway</surname> <given-names>H.</given-names></name> <name><surname>Kumari</surname> <given-names>M.</given-names></name> <name><surname>Brunner</surname> <given-names>E.</given-names></name> <name><surname>Malik</surname> <given-names>M.</given-names></name> <name><surname>Marmot</surname> <given-names>M.</given-names></name></person-group> (<year>2003</year>). <article-title>Effects of moderate and vigorous physical activity on heart rate variability in a British study of civil servants</article-title>. <source>Am. J. Epidemiol.</source> <volume>158</volume>, <fpage>135</fpage>&#x02013;<lpage>143</lpage>. <pub-id pub-id-type="doi">10.1093/aje/kwg120</pub-id><pub-id pub-id-type="pmid">12851226</pub-id></citation></ref>
<ref id="B76">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Robinson</surname> <given-names>O. J.</given-names></name> <name><surname>Frank</surname> <given-names>M. J.</given-names></name> <name><surname>Sahakian</surname> <given-names>B. J.</given-names></name> <name><surname>Cools</surname> <given-names>R.</given-names></name></person-group> (<year>2010a</year>). <article-title>Dissociable responses to punishment in distinct striatal regions during reversal learning</article-title>. <source>Neuroimage</source> <volume>51</volume>, <fpage>1459</fpage>&#x02013;<lpage>1467</lpage>. <pub-id pub-id-type="doi">10.1016/j.neuroimage.2010.03.036</pub-id><pub-id pub-id-type="pmid">20303408</pub-id></citation></ref>
<ref id="B77">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Robinson</surname> <given-names>O. J.</given-names></name> <name><surname>Standing</surname> <given-names>H. R.</given-names></name> <name><surname>DeVito</surname> <given-names>E. E.</given-names></name> <name><surname>Cools</surname> <given-names>R.</given-names></name> <name><surname>Sahakian</surname> <given-names>B. J.</given-names></name></person-group> (<year>2010b</year>). <article-title>Dopamine precursor depletion improves punishment prediction during reversal learning in healthy females but not males</article-title>. <source>Psychopharmacology</source> <volume>211</volume>, <fpage>187</fpage>&#x02013;<lpage>195</lpage>. <pub-id pub-id-type="doi">10.1007/s00213-010-1880-1</pub-id></citation></ref>
<ref id="B78">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Rodriguez</surname> <given-names>P. F.</given-names></name> <name><surname>Aron</surname> <given-names>A. R.</given-names></name> <name><surname>Poldrack</surname> <given-names>R. A.</given-names></name></person-group> (<year>2006</year>). <article-title>Ventral-striatal/nucleus-accumbens sensitivity to prediction errors during classification learning</article-title>. <source>Hum. Brain Mapp.</source> <volume>27</volume>, <fpage>306</fpage>&#x02013;<lpage>313</lpage>. <pub-id pub-id-type="doi">10.1002/hbm.20186</pub-id><pub-id pub-id-type="pmid">16092133</pub-id></citation></ref>
<ref id="B79">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Schmidt</surname> <given-names>R. E.</given-names></name> <name><surname>Gay</surname> <given-names>P.</given-names></name> <name><surname>D&#x00027;Acremont</surname> <given-names>M.</given-names></name> <name><surname>Van der Linden</surname> <given-names>M.</given-names></name></person-group> (<year>2008</year>). <article-title>A German adaptation of the UPPS impulsive behavior scale: psychometric properties and factor structure</article-title>. <source>Swiss J. Psychol.</source> <volume>67</volume>, <fpage>107</fpage>&#x02013;<lpage>112</lpage>. <pub-id pub-id-type="doi">10.1024/1421-0185.67.2.107</pub-id></citation></ref>
<ref id="B80">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Sch&#x000F6;nberg</surname> <given-names>T.</given-names></name> <name><surname>Daw</surname> <given-names>N. D.</given-names></name> <name><surname>Joel</surname> <given-names>D.</given-names></name> <name><surname>O&#x00027;Doherty</surname> <given-names>J. P.</given-names></name></person-group> (<year>2007</year>). <article-title>Reinforcement learning signals in the human striatum distinguish learners from nonlearners during reward-based decision making</article-title>. <source>J. Neurosci.</source> <volume>27</volume>, <fpage>12860</fpage>&#x02013;<lpage>12867</lpage>. <pub-id pub-id-type="doi">10.1523/JNEUROSCI.2496-07.2007</pub-id><pub-id pub-id-type="pmid">18032658</pub-id></citation></ref>
<ref id="B81">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Schrimpf</surname> <given-names>A.</given-names></name> <name><surname>Kube</surname> <given-names>J.</given-names></name> <name><surname>Neumann</surname> <given-names>J.</given-names></name> <name><surname>Horstmann</surname> <given-names>A.</given-names></name> <name><surname>Villringer</surname> <given-names>A.</given-names></name> <name><surname>Gaebler</surname> <given-names>M.</given-names></name></person-group> (<year>2017</year>). <article-title>Parasympathetic cardio-regulation during social interactions in individuals with obesity-The influence of negative body image</article-title>. <source>Cogn. Affect. Behav. Neurosci</source>. <volume>17</volume>, <fpage>330</fpage>&#x02013;<lpage>347</lpage>. <pub-id pub-id-type="doi">10.3758/s13415-016-0482-8</pub-id><pub-id pub-id-type="pmid">27905081</pub-id></citation></ref>
<ref id="B82">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Schroeder</surname> <given-names>E. B.</given-names></name> <name><surname>Liao</surname> <given-names>D.</given-names></name> <name><surname>Chambless</surname> <given-names>L. E.</given-names></name> <name><surname>Prineas</surname> <given-names>R. J.</given-names></name> <name><surname>Evans</surname> <given-names>G. W.</given-names></name> <name><surname>Heiss</surname> <given-names>G.</given-names></name></person-group> (<year>2003</year>). <article-title>Hypertension, blood pressure, and heart rate variability</article-title>. <source>Hypertension</source> <volume>42</volume>, <fpage>1106</fpage>&#x02013;<lpage>1111</lpage>. <pub-id pub-id-type="doi">10.1161/01.HYP.0000100444.71069.73</pub-id><pub-id pub-id-type="pmid">14581296</pub-id></citation></ref>
<ref id="B83">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Schultz</surname> <given-names>W.</given-names></name></person-group> (<year>2002</year>). <article-title>Getting formal with dopamine and reward</article-title>. <source>Neuron</source> <volume>36</volume>, <fpage>241</fpage>&#x02013;<lpage>263</lpage>. <pub-id pub-id-type="doi">10.1016/S0896-6273(02)00967-4</pub-id><pub-id pub-id-type="pmid">12383780</pub-id></citation></ref>
<ref id="B84">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Schultz</surname> <given-names>W.</given-names></name> <name><surname>Dayan</surname> <given-names>P.</given-names></name> <name><surname>Montague</surname> <given-names>P. R.</given-names></name></person-group> (<year>1997</year>). <article-title>A neural substrate of prediction and reward</article-title>. <source>Science</source> <volume>275</volume>, <fpage>1593</fpage>&#x02013;<lpage>1599</lpage>. <pub-id pub-id-type="doi">10.1126/science.275.5306.1593</pub-id><pub-id pub-id-type="pmid">9054347</pub-id></citation></ref>
<ref id="B85">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Schwarz</surname> <given-names>G. E.</given-names></name></person-group> (<year>1978</year>). <article-title>Estimating the dimension of a model</article-title>. <source>Ann. Stat.</source> <volume>6</volume>, <fpage>461</fpage>&#x02013;<lpage>464</lpage>. <pub-id pub-id-type="doi">10.1214/aos/1176344136</pub-id></citation></ref>
<ref id="B86">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Somsen</surname> <given-names>R. J. M.</given-names></name> <name><surname>Van der Molen</surname> <given-names>M. W.</given-names></name> <name><surname>Jennings</surname> <given-names>J. R.</given-names></name> <name><surname>van Beek</surname> <given-names>B.</given-names></name></person-group> (<year>2000</year>). <article-title>Wisconsin card sorting in adolescents: analysis of performance, response times and heart rate</article-title>. <source>Acta Psychol.</source> <volume>104</volume>, <fpage>227</fpage>&#x02013;<lpage>257</lpage>. <pub-id pub-id-type="doi">10.1016/S0001-6918(00)00030-5</pub-id><pub-id pub-id-type="pmid">10900707</pub-id></citation></ref>
<ref id="B87">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Stice</surname> <given-names>E.</given-names></name> <name><surname>Spoor</surname> <given-names>S.</given-names></name> <name><surname>Ng</surname> <given-names>J.</given-names></name> <name><surname>Zald</surname> <given-names>D. H.</given-names></name></person-group> (<year>2009</year>). <article-title>Relation of obesity to consummatory and anticipatory food reward</article-title>. <source>Physiol. Behav.</source> <volume>97</volume>, <fpage>551</fpage>&#x02013;<lpage>560</lpage>. <pub-id pub-id-type="doi">10.1016/j.physbeh.2009.03.020</pub-id><pub-id pub-id-type="pmid">19328819</pub-id></citation></ref>
<ref id="B88">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Strobel</surname> <given-names>A.</given-names></name> <name><surname>Beauducel</surname> <given-names>A.</given-names></name> <name><surname>Debener</surname> <given-names>S.</given-names></name> <name><surname>Brocke</surname> <given-names>B.</given-names></name></person-group> (<year>2001</year>). <article-title>Eine deutschsprachige Version des BIS/BAS-Fragebogens von Carver und White</article-title>. <source>Zeitschrift Differentielle Diagnostische Psychol.</source> <volume>22</volume>, <fpage>216</fpage>&#x02013;<lpage>227</lpage>. <pub-id pub-id-type="doi">10.1024//0170-1789.22.3.216</pub-id></citation></ref>
<ref id="B89">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Sutton</surname> <given-names>R. S.</given-names></name> <name><surname>Barto</surname> <given-names>A. G.</given-names></name></person-group> (<year>1998</year>). <source>Reinforcement Learning.</source> <publisher-loc>Cambridge</publisher-loc>: <publisher-name>MIT press</publisher-name>.</citation></ref>
<ref id="B90">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Symmonds</surname> <given-names>M.</given-names></name> <name><surname>Emmanuel</surname> <given-names>J. J.</given-names></name> <name><surname>Drew</surname> <given-names>M. E.</given-names></name> <name><surname>Batterham</surname> <given-names>R. L.</given-names></name> <name><surname>Dolan</surname> <given-names>R. J.</given-names></name></person-group> (<year>2010</year>). <article-title>Metabolic state alters economic decision making under risk in humans</article-title>. <source>PLoS ONE</source> <volume>5</volume>:<fpage>e11090</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pone.0011090</pub-id><pub-id pub-id-type="pmid">20585383</pub-id></citation></ref>
<ref id="B91">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Thayer</surname> <given-names>J. F.</given-names></name> <name><surname>Loerbroks</surname> <given-names>A.</given-names></name> <name><surname>Sternberg</surname> <given-names>E. M.</given-names></name></person-group> (<year>2011</year>). <article-title>Inflammation and cardiorespiratory control: the role of the vagus nerve</article-title>. <source>Respir. Physiol. Neurobiol.</source> <volume>178</volume>, <fpage>387</fpage>&#x02013;<lpage>394</lpage>. <pub-id pub-id-type="doi">10.1016/j.resp.2011.05.016</pub-id><pub-id pub-id-type="pmid">21642019</pub-id></citation></ref>
<ref id="B92">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Tzotzas</surname> <given-names>T.</given-names></name> <name><surname>Krassas</surname> <given-names>G. E.</given-names></name> <name><surname>Konstantinidis</surname> <given-names>T.</given-names></name> <name><surname>Bougoulia</surname> <given-names>M.</given-names></name></person-group> (<year>2000</year>). <article-title>Changes in lipoprotein(a) levels in overt and subclinical hypothyroidism before and during treatment</article-title>. <source>Thyroid</source> <volume>10</volume>, <fpage>803</fpage>&#x02013;<lpage>808</lpage>. <pub-id pub-id-type="doi">10.1089/thy.2000.10.803</pub-id><pub-id pub-id-type="pmid">11041458</pub-id></citation></ref>
<ref id="B93">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Van Der Veen</surname> <given-names>F. M.</given-names></name> <name><surname>Van Der Molen</surname> <given-names>M. W.</given-names></name> <name><surname>Crone</surname> <given-names>E. A.</given-names></name> <name><surname>Jennings</surname> <given-names>J. R.</given-names></name></person-group> (<year>2004</year>). <article-title>Phasic heart rate responses to performance feedback in a time production task: effects of information versus valence</article-title>. <source>Biol. Psychol</source>. <volume>65</volume>, <fpage>147</fpage>&#x02013;<lpage>161</lpage>. <pub-id pub-id-type="doi">10.1016/j.biopsycho.2003.07.003</pub-id><pub-id pub-id-type="pmid">14706437</pub-id></citation></ref>
<ref id="B94">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Vaschillo</surname> <given-names>E. G.</given-names></name> <name><surname>Vaschillo</surname> <given-names>B.</given-names></name> <name><surname>Buckman</surname> <given-names>J. F.</given-names></name> <name><surname>Nguyen-Louie</surname> <given-names>T.</given-names></name> <name><surname>Heiss</surname> <given-names>S.</given-names></name> <name><surname>Pandina</surname> <given-names>R. J.</given-names></name> <etal/></person-group>. (<year>2015</year>). <article-title>The effects of sighing on the cardiovascular system</article-title>. <source>Biol. Psychol.</source> <volume>106</volume>, <fpage>86</fpage>&#x02013;<lpage>95</lpage>. <pub-id pub-id-type="doi">10.1016/j.biopsycho.2015.02.007</pub-id><pub-id pub-id-type="pmid">25720947</pub-id></citation></ref>
<ref id="B95">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Volkow</surname> <given-names>N. D.</given-names></name> <name><surname>Wang</surname> <given-names>G. J.</given-names></name> <name><surname>Baler</surname> <given-names>R. D.</given-names></name></person-group> (<year>2011</year>). <article-title>Reward, dopamine and the control of food intake: implications for obesity</article-title>. <source>Trends Cogn. Sci</source>. <volume>15</volume>, <fpage>37</fpage>&#x02013;<lpage>46</lpage>. <pub-id pub-id-type="doi">10.1016/j.tics.2010.11.001</pub-id><pub-id pub-id-type="pmid">21109477</pub-id></citation></ref>
<ref id="B96">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>G. J.</given-names></name> <name><surname>Volkow</surname> <given-names>N. D.</given-names></name> <name><surname>Logan</surname> <given-names>J.</given-names></name> <name><surname>Pappas</surname> <given-names>N. R.</given-names></name> <name><surname>Wong</surname> <given-names>C. T.</given-names></name> <name><surname>Zhu</surname> <given-names>W.</given-names></name> <etal/></person-group>. (<year>2001</year>). <article-title>Brain dopamine and obesity</article-title>. <source>Lancet</source> <volume>357</volume>, <fpage>354</fpage>&#x02013;<lpage>357</lpage>. <pub-id pub-id-type="doi">10.1016/S0140-6736(00)03643-6</pub-id><pub-id pub-id-type="pmid">11210998</pub-id></citation></ref>
<ref id="B97">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Watkins</surname> <given-names>C. J. C. H.</given-names></name> <name><surname>Dayan</surname> <given-names>P.</given-names></name></person-group> (<year>1992</year>). <article-title>Technical note Q-learning</article-title>. <source>Mach. Learn.</source> <volume>8</volume>, <fpage>279</fpage>&#x02013;<lpage>292</lpage>. <pub-id pub-id-type="doi">10.1007/BF00992698</pub-id></citation></ref>
<ref id="B98">
<citation citation-type="book"><person-group person-group-type="author"><name><surname>Wechsler</surname> <given-names>D.</given-names></name></person-group> (<year>1987</year>). <source>Wechsler Memory Scale&#x02014;Revised manual</source>. <publisher-loc>San Antonio, TX</publisher-loc>: <publisher-name>Psychological Corporation</publisher-name>.</citation></ref>
<ref id="B99">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Weller</surname> <given-names>J. A.</given-names></name> <name><surname>Levin</surname> <given-names>I. P.</given-names></name> <name><surname>Bechara</surname> <given-names>A.</given-names></name></person-group> (<year>2009</year>). <article-title>Do individual differences in Iowa gambling task performance predict adaptive decision making for risky gains and losses?</article-title> <source>J. Clin. Exp. Neurophysiol.</source> <volume>32</volume>, <fpage>141</fpage>&#x02013;<lpage>150</lpage>. <pub-id pub-id-type="doi">10.1080/13803390902881926</pub-id><pub-id pub-id-type="pmid">19484643</pub-id></citation></ref>
<ref id="B100">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Weller</surname> <given-names>R. E.</given-names></name> <name><surname>Cook</surname> <given-names>E. W.</given-names></name> <name><surname>Avsar</surname> <given-names>K. B.</given-names></name> <name><surname>Cox</surname> <given-names>J. E.</given-names></name></person-group> (<year>2008</year>). <article-title>Obese women show greater delay discounting than healthy-weight women</article-title>. <source>Appetite</source> <volume>51</volume>, <fpage>563</fpage>&#x02013;<lpage>569</lpage>. <pub-id pub-id-type="doi">10.1016/j.appet.2008.04.010</pub-id><pub-id pub-id-type="pmid">18513828</pub-id></citation></ref>
<ref id="B101">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Whiteside</surname> <given-names>S. P.</given-names></name> <name><surname>Lynam</surname> <given-names>D. R.</given-names></name></person-group> (<year>2001</year>). <article-title>The Five factor model and impulsivity: using a structural model of personality to understand impulsivity</article-title>. <source>Pers. Individ. Dif.</source> <volume>30</volume>, <fpage>669</fpage>&#x02013;<lpage>689</lpage>. <pub-id pub-id-type="doi">10.1016/S0191-8869(00)00064-7</pub-id></citation></ref>
<ref id="B102">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Windham</surname> <given-names>B. G.</given-names></name> <name><surname>Fumagalli</surname> <given-names>S.</given-names></name> <name><surname>Ble</surname> <given-names>A.</given-names></name> <name><surname>Sollers</surname> <given-names>J. J.</given-names></name> <name><surname>Thayer</surname> <given-names>J. F.</given-names></name> <name><surname>Najjar</surname> <given-names>S. S.</given-names></name> <etal/></person-group>. (<year>2012</year>). <article-title>The Relationship between heart rate variability and adiposity differs for central and overall adiposity</article-title>. <source>J. Obes</source>. <volume>2012</volume>:<fpage>149516</fpage>. <pub-id pub-id-type="doi">10.1155/2012/149516</pub-id><pub-id pub-id-type="pmid">22649714</pub-id></citation></ref>
<ref id="B103">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Wittmann</surname> <given-names>B. C.</given-names></name> <name><surname>D&#x00027;Esposito</surname> <given-names>M.</given-names></name></person-group> (<year>2015</year>). <article-title>Levodopa administration modulates striatal processing of punishment-associated items in healthy participants</article-title>. <source>Psychopharmacology</source> <volume>232</volume>, <fpage>135</fpage>&#x02013;<lpage>144</lpage>. <pub-id pub-id-type="doi">10.1007/s00213-014-3646-7</pub-id><pub-id pub-id-type="pmid">24923987</pub-id></citation></ref>
<ref id="B104">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zahorska-Markiewicz</surname> <given-names>B.</given-names></name> <name><surname>Kuagowska</surname> <given-names>E.</given-names></name> <name><surname>Kucio</surname> <given-names>C.</given-names></name> <name><surname>Klein</surname> <given-names>M.</given-names></name></person-group> (<year>1993</year>). <article-title>Heart rate variability in obesity</article-title>. <source>Int. J. Obes. Relat. Metab. Disord</source>. <volume>17</volume>, <fpage>21</fpage>&#x02013;<lpage>23</lpage>. <pub-id pub-id-type="pmid">8383637</pub-id></citation></ref>
<ref id="B105">
<citation citation-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>Z.</given-names></name> <name><surname>Manson</surname> <given-names>K. F.</given-names></name> <name><surname>Schiller</surname> <given-names>D.</given-names></name> <name><surname>Levy</surname> <given-names>I.</given-names></name></person-group> (<year>2014</year>). <article-title>Impaired associative learning with food rewards in obese women</article-title>. <source>Curr. Biol.</source> <volume>24</volume>, <fpage>1731</fpage>&#x02013;<lpage>1736</lpage>. <pub-id pub-id-type="doi">10.1016/j.cub.2014.05.075</pub-id><pub-id pub-id-type="pmid">25042588</pub-id></citation></ref>
</ref-list>
<fn-group>
<fn fn-type="financial-disclosure"><p><bold>Funding.</bold> This work was supported by the German Federal Ministry of Education and Research (FKZ: 01EO1001; LK, JK, AV, and JN) and the German Research Foundation (SFB 1052 Obesity mechanisms; AV and JN).</p>
</fn>
</fn-group>
</back>
</article>