<?xml version="1.0" ?><!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Archiving and Interchange DTD v1.3 20210610//EN"  "JATS-archivearticle1-mathml3.dtd"><article xmlns:ali="http://www.niso.org/schemas/ali/1.0/" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article" dtd-version="1.3" xml:lang="en">
<front>
<journal-meta>
<journal-id journal-id-type="nlm-ta">elife</journal-id>
<journal-id journal-id-type="publisher-id">eLife</journal-id>
<journal-title-group>
<journal-title>eLife</journal-title>
</journal-title-group>
<issn publication-format="electronic" pub-type="epub">2050-084X</issn>
<publisher>
<publisher-name>eLife Sciences Publications, Ltd</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">103363</article-id>
<article-id pub-id-type="doi">10.7554/eLife.103363</article-id>
<article-id pub-id-type="doi" specific-use="version">10.7554/eLife.103363.1</article-id>
<article-version-alternatives>
<article-version article-version-type="publication-state">reviewed preprint</article-version>
<article-version article-version-type="preprint-version">1.2</article-version>
</article-version-alternatives>
<article-categories><subj-group subj-group-type="heading">
<subject>Neuroscience</subject>
</subj-group>
</article-categories>
<title-group>
<article-title>Multiple and subject-specific roles of uncertainty in reward-guided decision-making</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<contrib-id contrib-id-type="orcid">http://orcid.org/0000-0002-0547-717X</contrib-id>
<name>
<surname>Paunov</surname>
<given-names>Alexander</given-names>
</name>
<xref ref-type="aff" rid="a1">1</xref>
<xref ref-type="aff" rid="a2">2</xref>
<email>alexander.paunov@cea.fr</email>
</contrib>
<contrib contrib-type="author">
<name>
<surname>L’Hôtellier</surname>
<given-names>Maëva</given-names>
</name>
<xref ref-type="aff" rid="a1">1</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Guo</surname>
<given-names>Dalin</given-names>
</name>
<xref ref-type="aff" rid="a3">3</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>He</surname>
<given-names>Zoe</given-names>
</name>
<xref ref-type="aff" rid="a3">3</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Yu</surname>
<given-names>Angela</given-names>
</name>
<xref ref-type="aff" rid="a3">3</xref>
<xref ref-type="aff" rid="a4">4</xref>
</contrib>
<contrib contrib-type="author" corresp="yes">
<contrib-id contrib-id-type="orcid">http://orcid.org/0000-0002-6992-678X</contrib-id>
<name>
<surname>Meyniel</surname>
<given-names>Florent</given-names>
</name>
<xref ref-type="aff" rid="a1">1</xref>
<xref ref-type="aff" rid="a2">2</xref>
<email>florent.meyniel@cea.fr</email>
</contrib>
<aff id="a1"><label>1</label><institution-wrap><institution-id institution-id-type="ror">https://ror.org/03n15ch10</institution-id><institution>INSERM-CEA Cognitive Neuroimaging Unit (UNICOG), NeuroSpin Center, CEA Paris-Saclay, Gif-sur-Yvette, France Université de Paris</institution></institution-wrap>, <city>Paris</city>, <country>France</country></aff>
<aff id="a2"><label>2</label><institution-wrap><institution-id institution-id-type="ror">https://ror.org/02kxjxy06</institution-id><institution>Institut de Neuromodulation, GHU Paris, Psychiatrie et Neurosciences, Centre Hospitalier Sainte-Anne, Pôle Hospitalo-Universitaire 15, Université Paris Cité</institution></institution-wrap>, <city>Paris</city>, <country>France</country></aff>
<aff id="a3"><label>3</label><institution-wrap><institution-id institution-id-type="ror">https://ror.org/0168r3w48</institution-id><institution>Department of Cognitive Science, University of California San Diego</institution></institution-wrap>, <city>San Diego</city>, <country>USA</country></aff>
<aff id="a4"><label>4</label><institution-wrap><institution-id institution-id-type="ror">https://ror.org/05n911h24</institution-id><institution>Centre for Cognitive Science &amp; Hessian AI Center, Technical University of Darmstadt</institution></institution-wrap>, <city>Darmstadt</city>, <country>Germany</country></aff>
</contrib-group>
<contrib-group content-type="section">
<contrib contrib-type="editor">
<name>
<surname>Diaconescu</surname>
<given-names>Andreea Oliviana</given-names>
</name>
<role>Reviewing Editor</role>
<aff>
<institution-wrap>
<institution>University of Toronto</institution>
</institution-wrap>
<city>Toronto</city>
<country>Canada</country>
</aff>
</contrib>
<contrib contrib-type="senior_editor">
<name>
<surname>Frank</surname>
<given-names>Michael J</given-names>
</name>
<role>Senior Editor</role>
<aff>
<institution-wrap>
<institution>Brown University</institution>
</institution-wrap>
<city>Providence</city>
<country>United States of America</country>
</aff>
</contrib>
</contrib-group>
<pub-date date-type="original-publication" iso-8601-date="2024-12-06">
<day>06</day>
<month>12</month>
<year>2024</year>
</pub-date>
<volume>13</volume>
<elocation-id>RP103363</elocation-id>
<history>
<date date-type="sent-for-review" iso-8601-date="2024-09-23">
<day>23</day>
<month>09</month>
<year>2024</year>
</date>
</history>
<pub-history>
<event>
<event-desc>Preprint posted</event-desc>
<date date-type="preprint" iso-8601-date="2024-09-12">
<day>12</day>
<month>09</month>
<year>2024</year>
</date>
<self-uri content-type="preprint" xlink:href="https://doi.org/10.1101/2024.03.27.587016"/>
</event>
</pub-history>
<permissions>
<copyright-statement>© 2024, Paunov et al</copyright-statement>
<copyright-year>2024</copyright-year>
<copyright-holder>Paunov et al</copyright-holder>
<ali:free_to_read/>
<license xlink:href="https://creativecommons.org/licenses/by/4.0/">
<ali:license_ref>https://creativecommons.org/licenses/by/4.0/</ali:license_ref>
<license-p>This article is distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution License</ext-link>, which permits unrestricted use and redistribution provided that the original author and source are credited.</license-p>
</license>
</permissions>
<self-uri content-type="pdf" xlink:href="elife-preprint-103363-v1.pdf"/>
<abstract>
<title>Abstract</title><p>Decision-making in noisy, changing, and partially observable environments entails a basic tradeoff between immediate reward and longer-term information gain, known as the exploration-exploitation dilemma. Computationally, an effective way to balance this tradeoff is by leveraging uncertainty to guide exploration. Yet, in humans, empirical findings are mixed, from suggesting uncertainty-seeking to indifference and avoidance. In a novel bandit task that better captures uncertainty-driven behavior, we find multiple roles for uncertainty in human choices. First, stable and psychologically meaningful individual differences in uncertainty preferences actually range from seeking to avoidance, which can manifest as null group-level effects. Second, uncertainty modulates the use of basic decision heuristics that imperfectly exploit immediate rewards: a repetition bias and win-stay-lose-shift heuristic. These heuristics interact with uncertainty, favoring heuristic choices under higher uncertainty. These results, highlighting the rich and varied structure of reward-based choice, are a step to understanding its functional basis and dysfunction in psychopathology.</p>
</abstract>
<custom-meta-group>
<custom-meta specific-use="meta-only">
<meta-name>publishing-route</meta-name>
<meta-value>prc</meta-value>
</custom-meta>
</custom-meta-group>
</article-meta>
<notes>
<notes notes-type="competing-interest-statement">
<title>Competing Interest Statement</title><p>The authors have declared no competing interest.</p></notes>
<fn-group content-type="summary-of-updates">
<title>Summary of Updates:</title>
<fn fn-type="update"><p>Fixing the poor figure resolution in the original upload.
Also minor revisions such as adding references and minor clarifications in the text and figure legends, fixing figure references.</p></fn>
</fn-group>
</notes>
</front>
<body>
<sec id="s1">
<title>Introduction</title>
<p>Agents making sequential choices under uncertainty face a pervasive tradeoff between learning about their environment and capitalizing on known rewards. This “exploration-exploitation” dilemma applies to bumblebees deciding where to forage <sup><xref ref-type="bibr" rid="c1">1</xref></sup>, children making causal inferences <sup><xref ref-type="bibr" rid="c2">2</xref></sup>, adults deliberating economic choices <sup><xref ref-type="bibr" rid="c3">3</xref></sup>, and algorithms running stochastic optimization <sup><xref ref-type="bibr" rid="c4">4</xref></sup>. Under most ecological conditions, there is either no known or no practical optimal solution to this tradeoff, in the sense that it would provide theoretical guarantees that long-term rewards are maximized <sup><xref ref-type="bibr" rid="c5">5</xref></sup>. Such ecological conditions arise when observations are noisy, reward contingencies change over time, the decision problem has a long or indefinite time horizon, and/or the decision space is large, among others. A broad question that arises, then, is what are the algorithms that organisms, and humans in particular, use to resolve the explore-exploit tradeoff in practice and make sensible decisions?</p>
<p>Based on Gittins’ <sup><xref ref-type="bibr" rid="c6">6</xref></sup> original finding that an optimally exploring agent (in a particular simple and mathematically well-characterized setting) ought to confer an “uncertainty bonus” to options in addition to their expected reward values <sup><xref ref-type="bibr" rid="c5">5</xref></sup>, a number of modeling studies have explored the role of uncertainty in helping humans balance the tradeoff between exploration and exploitation. Some studies have found support for these or similar strategies in humans <sup><xref ref-type="bibr" rid="c7">7</xref>–<xref ref-type="bibr" rid="c12">12</xref></sup>, but there remain outstanding puzzles. First, results are rather mixed, with some other studies reporting uncertainty-avoidance effects <sup><xref ref-type="bibr" rid="c13">13</xref>,<xref ref-type="bibr" rid="c14">14</xref></sup>, or no uncertainty-based exploration effects <sup><xref ref-type="bibr" rid="c15">15</xref>,<xref ref-type="bibr" rid="c16">16</xref></sup>. This raises the question of what may cause heterogeneity in the use of uncertainty-based exploration strategies. Previous work has found substantial individual differences in the use of uncertainty in exploration, linked to psychological measures including impulsivity and anxiety <sup><xref ref-type="bibr" rid="c15">15</xref>,<xref ref-type="bibr" rid="c17">17</xref>–<xref ref-type="bibr" rid="c20">20</xref></sup>. Here, we tested the hypothesis that there is considerable heterogeneity of uncertainty effects that reflects differences across individuals. Such heterogeneity could explain mixed results of previous studies, especially in small or non-random samples.</p>
<p>Second, some apparent effects of uncertainty can be accounted for by more basic strategies than incorporating uncertainty in decisions trial-by-trial. In many experimental contexts, such as ones involving stationary rewards and short time horizons <sup><xref ref-type="bibr" rid="c7">7</xref>–<xref ref-type="bibr" rid="c9">9</xref>,<xref ref-type="bibr" rid="c11">11</xref>,<xref ref-type="bibr" rid="c17">17</xref></sup>, simple proxies for uncertainty, such as familiarity,novelty, or count-based heuristics may drive exploratory choices <sup><xref ref-type="bibr" rid="c7">7</xref>,<xref ref-type="bibr" rid="c13">13</xref>,<xref ref-type="bibr" rid="c21">21</xref>,<xref ref-type="bibr" rid="c22">22</xref></sup>. Other basic heuristics, which do not require integrating information over many trials, such as a repetition bias or a win-stay-lose-shift strategy, may also mimic uncertainty effects, if not controlled for <sup><xref ref-type="bibr" rid="c23">23</xref>,<xref ref-type="bibr" rid="c24">24</xref></sup>. For example, a tendency to switch upon encountering a large negative deviation from expectation to an expectedly lower value option can appear mistakenly as uncertainty-seeking behavior. Likewise, repetition bias, a form of perseveration (the tendency to repeat an option even when this does not maximize immediate reward), can appear uncertainty-driven, depending on how the decision policy is parameterized. Such basic heuristics are ubiquitous in human and animal decision-making <sup><xref ref-type="bibr" rid="c25">25</xref>–<xref ref-type="bibr" rid="c35">35</xref></sup>, and may reflect cognitive constraints on decision-making <sup><xref ref-type="bibr" rid="c36">36</xref>–<xref ref-type="bibr" rid="c40">40</xref></sup>. This raises the question of whether uncertainty effects are still observed when these heuristics are taken into account. It also raises the possibility that uncertainty may play further roles in human decision-making by interacting with such basic heuristics, i.e., by increasing or decreasing the degree to which they are used in a context-specific manner. In the present study, we set out to test potential roles of multiple decision factors, by identifying the contributions of these basic heuristics, repetition and wins-stay-lose-shift, alongside uncertainty and expected reward in guiding sequential choices, and to investigate their potential interactions.</p>
<p>We used a novel task design that deconfounds uncertainty from alternatives and expected reward. To anticipate our results, we find that (1) multiple factors in addition to uncertainty contribute to subjects’ choices; (2) uncertainty has heterogeneous effects on choices across subjects, and (3) these differences are stable across testing days within an individual; (4) repetition bias and win-stay-lose-shift tendencies contribute to choices pervasively across subjects; and, finally, (5) uncertainty modulates the use of these heuristics.</p>
</sec>
<sec id="s2">
<title>Results</title>
<sec id="s2a">
<label>3.1</label><title>A novel explore-exploit task to characterize the role(s) of uncertainty</title>
<p>To study the roles of uncertainty in choices within a broader set of factors implicated in reward-based decisions, we adopt a novel variant of a two-armed bandit task (<xref rid="fig1" ref-type="fig">Fig. 1A</xref>). Subjects make a long sequence of choices (96 per block) between two options delivering rewards between 1 and 100 points with independently evolving reward distributions. They are only shown the reward of the chosen option, creating an explore-exploit tension. The need to explore is maintained by abrupt uncued shifts in reward levels (one out of three possibilities), termed changepoints (which occur independently for the two options), and noisy observations on each trial (drawn from Gaussians around a mean reward level). Cued low and high levels of observation noise are used across halves of a block to make changepoints more or less detectable, respectively, and to add further variation in estimation uncertainty (i.e. the observer’s uncertainty about the latent reward level; thereafter simply termed “uncertainty”). Expected reward and uncertainty tend to be negatively correlated in bandit tasks with partial information, because more rewarding options are (rightly) sampled more often <sup><xref ref-type="bibr" rid="c11">11</xref></sup>. To decorrelate them, we interleave forced choice periods with free choices (see Methods for more details). A notable advantage of using long, nonstationary sequences, is that neither novelty nor familiarity <sup><xref ref-type="bibr" rid="c7">7</xref>,<xref ref-type="bibr" rid="c13">13</xref>,<xref ref-type="bibr" rid="c21">21</xref></sup> can account for exploratory choices, providing a more compelling test for genuinely uncertainty-based components of choice. Finally, subjects are occasionally asked to make explicit reports of reward level estimates and uncertainty / confidence about these guesses, to provide a choice-independent and direct (model-agnostic) measure of the factors that may guide their choices.</p>
<fig id="fig1" position="float" orientation="portrait" fig-type="figure">
<label>Figure 1.</label>
<caption><title>Task and model</title>
<p><bold>A.</bold> On each trial subjects choose between two options and observe the outcome of the selected option. On forced trials, only the circled option can be selected (here, the purple one). <bold>B.</bold> Every 16 trials on average (SD=4.85) trials, subjects are asked to report their estimate of option values (black outline). Example of reward level estimate (left) and confidence report (right); only one option is shown here but both options are queried sequentially <bold>C.</bold> Example block. The latent mean reward levels (solid lines) shifted abruptly between 3 reward levels at uncued changepoints, independently for each option. Outcomes (dots) were drawn from Gaussians around the mean level with either low or high standard deviation. Each block was split into low-noise and high-noise halves corresponding to the low/high standard deviation of sampled outcomes. Subjects were instructed about noise levels, which were cued using a single or double circle around the fixation dot. Periods of 8 free trials alternated with periods of 4 forced trials (denoted by white and gray rectangles, circles and crosses, respectively). <bold>D.</bold> Posterior probability of an option following a single observation computed by the Bayesian ideal observer model. Model-based quantities were derived from the posterior: Expected reward (ER) is the sum of the reward levels weighted by their probabilities. Estimation uncertainty (EU) is 1 minus the probability of the maximum a posteriori (MAP) reward level.</p></caption>
<graphic xlink:href="587016v2_fig1.tif" mime-subtype="tiff" mimetype="image"/>
</fig>
<p>We use a Bayesian ideal observer as a learning model to derive candidate decision factors. This learning model estimates the posterior probability of the latent reward levels for each option given the choices and outcomes observed by subjects. Notably, unlike point-estimates of reward expectation obtained from standard delta-rule models, the Bayesian learner also provides trial-by-trial uncertainty estimates about the latent reward level, which can be used as a proxy for subjects’ uncertainty <sup><xref ref-type="bibr" rid="c41">41</xref></sup>. This formalization of uncertainty corresponds to <italic>estimation uncertainty</italic>, which depends both on the noisiness of the observations (variously known as risk, outcome variability, irreducible or expected uncertainty; <sup><xref ref-type="bibr" rid="c14">14</xref>,<xref ref-type="bibr" rid="c42">42</xref>,<xref ref-type="bibr" rid="c43">43</xref></sup>) and the sampling history. The decision factors that we derive from the learning model are (in addition to a <italic>repetition bias</italic>): <italic>immediate reward</italic>, defined as the relative expected reward (ΔER; difference between the two options); the <italic>relative and total uncertainties</italic> (ΔEU, EUt; corresponding respectively to the difference and sum across the two options), both of which have previously been implicated in uncertainty-based exploration <sup><xref ref-type="bibr" rid="c8">8</xref>,<xref ref-type="bibr" rid="c9">9</xref>,<xref ref-type="bibr" rid="c15">15</xref></sup>; a pair of factors capturing a <italic>win-stay-lose-shift heuristic</italic> (PE, UPE; the effect of signed and unsigned prediction errors on choices capture respectively the win-stay-lose-shift behavior and its potential asymmetry between the “win” and “loss” domains); and interactions between the uncertainties and the remaining factors (see Methods).</p>
<p>Our analysis strategy is as follows. Two prerequisites are, first, to validate that the Bayesian learning model accounts for subjects’ choices and reports, and second, to select the most appropriate form of choice stochasticity in the decision model (softmax or purely random decision noise). The learning and decision (i.e., policy) components are considered sequentially here for exposition, but in practice they are necessarily modeled jointly. The learning model can only be evaluated under a particular policy, and conversely, the definitions of the variables that comprise the decision policy depend on the learning model. Throughout this paper, we model the choice to repeat the previous decision or switch (rather than choosing option A vs B), because it provides a natural way to capture repetition bias, which we find to be a major contributor to choices in the task (see Results).</p>
<p>With these prerequisite modeling choices settled, our first main goal is to determine which factors contribute to choices via a model selection procedure. We then characterize the contributions of the main drivers of choice in addition to immediate reward, focusing first on the uncertainty-based factors and then on the basic choice heuristics, repetition bias and win-stay-lose-shift. Again, these sets of factors are considered sequentially for exposition, but they are modeled jointly, to determine each factor’s contribution while controlling for the rest. To test the hypothesis that variability in uncertainty effects across subjects reflects stable and psychologically meaningful individual differences, we assess the stability of uncertainty coefficients across testing sessions and their correlations with psychometric scales. To test the hypothesis that uncertainty modulates the use of the other heuristics, we ask whether and how uncertainty interacts with the remaining decision factors.</p>
</sec>
<sec id="s2b">
<label>3.2</label><title>Task performance</title>
<p>Firstly, we verified the subjects understood and were engaged in the task. Out of 59 subjects, only 3 (5%) did not perform significantly above chance (49.5 points, SD=0.56). These 3 subjects were excluded from further analysis (the main conclusions remain unchanged when they are included). For the remaining 56 subjects in the final sample, the average reward earned was 54.0, SD = 1.16; range 51.2 - 55.6 points. This performance was on average slightly but reliably lower than simulated optimal performance under the Bayesian learning model and the selected decision model presented below (see <xref rid="s2e" ref-type="sec">Section 3.5</xref>) (M = 55.1; SD = 0.05, p=10<sup>−8</sup>). The overall commensurate performance in subjects and in simulation suggests that the model adequately captures subjects’ performance in the task.</p>
</sec>
<sec id="s2c">
<label>3.3</label><title>A Bayesian model of learning accounts for choices and reports</title>
<sec id="s2c1">
<title>Choices</title>
<p>It is necessary to assume a learning model to study the latent factors that guide choices and their values on a trial-by-trial basis. A Bayesian model has an <italic>a priori</italic> advantage for studying the roles of uncertainty in decisions: it provides trial-by-trial uncertainty estimates, which can be used as a proxy of subjects’ uncertainty.</p>
<p>However, simpler reinforcement learning models, such as the Rescorla-Wagner (RW) delta rule model, are commonly used <sup><xref ref-type="bibr" rid="c44">44</xref>–<xref ref-type="bibr" rid="c47">47</xref></sup>. In contrast to a Bayesian model, this standard RW model only provides point estimates (no estimate of uncertainty), and assumes subjects do not use a learning procedure that is sensitive to the generative structure of the task. The standard version of RW only updates the expected reward of the option, for which there is a prediction error (i.e. the selected option here). To improve the ability of RW to better account for subjects’ choices, we augment it with updates of the unobserved option toward the average reward level, a feature that is also present in the Bayesian model (see Methods). The augmented RW model provided a better fit to choices than the standard RW (cross-validated average choice likelihoods for augmented RW: p<sub>cv</sub>(choice) = 0.702; for standard RW: p<sub>cv</sub>(choice) = 0.694; paired difference: 0.007, SE=0.0007, <italic>t</italic>(55) = 9.95, <italic>p</italic> = 10<sup>−15</sup>, Cohen’s <italic>d</italic> = 1.34). We find the Bayesian learning model accounts for subjects’ choices better than this augmented RW model (for the Bayesian model: p<sub>cv</sub>(choice) = 0.708; for RW: p<sub>cv</sub>(choice) = 0.702; paired difference: 0.007, SE=0.002, <italic>t</italic>(55) = 4.57, <italic>p</italic> = 10<sup>−6</sup>, Cohen’s <italic>d</italic> = 0.64; see Methods).</p>
</sec>
<sec id="s2c2">
<title>Reports</title>
<p>The inclusion of occasional explicit reports of the inferred mean reward levels and confidence permits an independent test of whether subjects represent value and uncertainty, and whether these representations resemble those of the Bayesian learner. Subjects’ guesses of the latent reward level match the true generative reward level and even more the ideal observer’s maximum a posteriori estimate well above chance (<xref rid="fig2" ref-type="fig">Fig. 2B</xref>) (fraction of reports matching generative values: 0.58, SE=0.012; matching the ideal observer: 0.61, SE=0.015; comparison to chance level (0.33): <italic>t</italic>(55)&gt;19.20, <italic>p&lt;</italic>10<sup>−27</sup>, Cohen’s <italic>d</italic> = 5.63; paired difference: 0.033, <italic>t</italic>(55) = 4.16; <italic>p</italic> =0.0001, Cohen’s <italic>d</italic> = 0.57), suggesting that the Bayesian latent reward estimates capture subjects’ subjective reward estimates.</p>
<fig id="fig2" position="float" orientation="portrait" fig-type="figure">
<label>Figure 2.</label>
<caption><title>Choice performance and reports.</title>
<p><bold>A</bold>. Average performance (i.e. payoff) across sessions. Oracle performance is the expected payoff when selecting the option with highest (unknown) latent value on each trial. Chance performance is the expected payoff when selecting randomly and evenly both options. <bold>B</bold>. Fraction of reports matching the Bayesian model estimate of the reward level or matching of the (unknown) latent reward level. <bold>C</bold>. Average subject confidence for bins of Bayesian model confidence (with volatility fitted to subjects’ choices) about the latent reward level (1-EU). In all panels: dots correspond to subjects, error bars to SEM.</p></caption>
<graphic xlink:href="587016v2_fig2.tif" mime-subtype="tiff" mimetype="image"/>
</fig>
<p>We also find subjects’ reported confidence reliably correlates with the Bayesian learner confidence (quantified as 1-EU) on the same trial (<xref rid="fig2" ref-type="fig">Fig. 2C</xref>): Pearson’s r<sub>mean</sub> = 0.30, SE = 0.02, <italic>t</italic>(55) = 14.60, <italic>p</italic> = 10<sup>−19</sup>, Cohen’s <italic>d</italic> = 1.97. At the individual level, this correlation is significant at <italic>p</italic> &lt; 0.05 in 77% of subjects (43 of 56). This suggests that subjects track uncertainty, and do so in a way that is similar to the Bayesian model. These results converge with evidence from behavioral modeling that the Bayesian model is a suitable model of subjects’ learning in the task.</p>
</sec>
<sec id="s2c3">
<title>Fitting volatility</title>
<p>Learning in the Bayesian model depends on the assumed volatility (probability of a changepoint on each trial). Analysis of this volatility parameter, fitted to each subject, provides convergent support that the Bayesian model is a useful approximation of subjects’ learning. Prior work suggests that humans tend to overestimate volatility <sup><xref ref-type="bibr" rid="c20">20</xref>,<xref ref-type="bibr" rid="c48">48</xref>–<xref ref-type="bibr" rid="c50">50</xref></sup>. Consistent with this, the volatility parameter fitted in each subject was on average higher than the generative volatility (vol<sub>fitted</sub>= 0.16, SE=0.01, <italic>t</italic>(55) = 11.33, <italic>p</italic> = 10<sup>−15</sup>; vol<sub>generative</sub> = 0.042, one-sample t-test against vol<sub>generative</sub>; Cohen’s <italic>d</italic> = 1.12), and models with volatility as a subject-specific free parameter outperformed equivalent models using the generative volatility (average difference in choice likelihood across models Δp<sub>cv</sub> = 0.02, SE = 0.002, <italic>t</italic>(55) = 9.83, <italic>p</italic> = 10<sup>−12</sup>, Cohen’s <italic>d</italic> = 1.31). Therefore, all reported results are from models with fitted volatility.</p>
<p>Further validating the Bayesian learning model, we compare the correlation of subjects’ explicit confidence reports to the model confidence with (<xref rid="fig2" ref-type="fig">Fig. 2C</xref>) vs without (not shown) fitting volatility. The confidence estimates derived from the model with fitted volatility were more strongly correlated with subjects’ ratings than those derived from the generative volatility model: Δr = 0.03, SE=0.009, r<sub>fitted_vol</sub> = 0.30, r<sub>generative_vol</sub> = 0.26, <italic>t</italic>(55) = 3.89, <italic>p</italic> = 10<sup>−4</sup>, Cohen’s <italic>d</italic> = 0.53. The Bayesian estimates of the posterior reward level with the generative vs. fitted volatility did not differ in how well they match the explicit reports (p(match)<sub>fitted_vol</sub> = 0.613; p(match)<sub>generative_vol</sub> = 0.618, <italic>n.s.</italic>). This may be due to lower sensitivity of this measure, relative to correlations with confidence ratings. These results suggest that by estimating the subjects “presumed volatility” from choices, the Bayesian learner provides an even better approximation to the learning process, which is reflected in a closer match of the model to subjects’ subjective confidence.</p>
</sec>
</sec>
<sec id="s2d">
<label>3.4</label><title>Choice stochasticity is better captured by a reward-guided than a fully random policy</title>
<p>Models of choices typically assume some degree of choice randomness to handle residual, unexplained variance in choices. It is relevant to distinguish between two forms of randomness in the context of exploration. Epsilon-greedy decision models postulate that choices either strictly maximize the decision value or are made randomly with a probability epsilon on each trial. The random component of such a model captures completely random exploration. In contrast, softmax decision models postulate that choices maximize the decision value more often when the estimated reward value difference is larger between options. This random component corresponds to another form of exploration that is not completely random, but inversely related to the value difference.</p>
<p>To determine which form of choice randomness better accounts for our data, we compare cross-validated model fits with a softmax policy and an epsilon greedy policy, for the model including only a repetition bias and expected reward. The softmax policy provides a better fit to choices than epsilon greedy: base model (i.e., the model including all main effects, and no interactions): softmax p(choice)<sub>cv</sub> = 0.708, ε-greedy p(choice)<sub>cv</sub> = 0.650; Δp<sub>cv</sub> = 0.058; SE = 0.007, <italic>t</italic>(55) = 7.99, <italic>p</italic> = 10<sup>−10</sup>, Cohen’s <italic>d</italic> = 1.08. Our interpretation of this result is that subjects’ propensity to choose the higher-valued option depends on the value difference. All reported results therefore use the softmax model.</p>
</sec>
<sec id="s2e">
<label>3.5</label><title>Multiple factors contribute to choices, with a dominant effect of expected reward</title>
<p>Having validated the Bayesian learning model and selected softmax as the model of choice randomness, our first main goal is to determine which factors drive behavior in the task. Subjects were instructed that their goal was to maximize points earned in the task, and the observed task performance (<xref rid="s2a" ref-type="sec">Section 3.1</xref>) already implies that immediate reward is a strong driver of choices since subjects are, on average, more likely to choose the more rewarding option. Our model selection strategy is therefore to consider a baseline model, consisting of the immediate reward term and a repetition bias (as the intercept), and to ask what additional factors significantly improve the cross-validated model fit, including the relative and total uncertainty (ΔEU and EUt, respectively capturing uncertainty-based exploration), and two prediction error terms (signed and unsigned PE, capturing the win-stay-lose-shift heuristic). We iteratively add all combinations of one to four factors, for a total of 16 possible combinations. Model pairs which only differ in the inclusion or exclusion of a single factor showed that each of the four additional factors significantly contributes to choices (<xref rid="fig2" ref-type="fig">Fig. 2A</xref>). To ensure that the parameters of the full model are distinguishable and that choices generated with a given model are best fit by the same model, we performed parameter and model recovery (<xref rid="figs1" ref-type="fig">Fig. S1</xref> and <xref rid="figs2" ref-type="fig">S2</xref>).</p>
<p>A strong test for the individual contributions of each factor are single-factor ablations from the full model, testing for the marginal contribution of a given factor while accounting for the rest (<xref rid="fig2" ref-type="fig">Fig. 2A</xref>, bottom row). The respective unique additional contributions of each factor in this test are Δp<sub>cv</sub>(choice): ΔEU = 0.11% (SE = 0.03, <italic>t</italic>(55) = 4.14, <italic>p</italic> = 10<sup>−4</sup>, Cohen’s <italic>d</italic> = 0.56); EUt = 0.07%, SE = 0.03, <italic>t</italic>(55) = 2.67, <italic>p</italic> = 10<sup>−3</sup>, Cohen’s <italic>d</italic> = 0.36); PE = 0.47%, SE=0.11, <italic>t</italic>(55) = 4.15, <italic>p</italic> = 10<sup>−4</sup>, Cohen’s <italic>d</italic> = 0.56, UPE = 0.19% (SE=0.04, <italic>t</italic>(55) = 4.65, <italic>p</italic> = 10<sup>−4</sup>, Cohen’s <italic>d</italic> = 0.63). Together, these four factors contribute an extra percentage-point (1%) to the cross-validated choice probability (p<sub>cv</sub>(choice) in the full model: 0.718, in the base model: 0.708). This should be compared to the respective unique additional contribution of expected reward (7.28%, SE = 0.45, <italic>t</italic>(55) = 16.26, <italic>p</italic> = 10<sup>−21</sup>, Cohen’s <italic>d</italic> = 2.17) and of the repetition bias (5.72%, SE = 0.71, <italic>t</italic>(55) = 8.05, <italic>p</italic> = 10<sup>−10</sup>, Cohen’s <italic>d</italic> = 1.08) in the full model. The above-chance performance of the model without the dominant factors (ablate ΔER: p(choice)<sub>cv</sub> = 0.65; ablate repetition bias: p(choice)<sub>cv</sub> = 0.66; ablate both: p(choice)<sub>cv</sub> = 0.59) reflects shared predictive variance among decision factors.</p>
<p>Model comparison within subjects indicates that the full model is the most frequent best model at the subject-level, followed by the two models that differ from the full model by lacking only either ΔEU or EUt (<xref rid="figs3" ref-type="fig">Fig. S3</xref>). This is an important result that rules out that the full model is the best at the group level because it is the only one that can accommodate the implication of different decision factors among subjects. Instead, it is the best model because many factors (including uncertainty) contribute to choices within subjects.</p>
<p>We also note that ignoring the heuristic factors (repetition bias, or PE and UPE) from the model of subjects’ choices inflates the estimate of ΔEU (<xref rid="figs4" ref-type="fig">Fig. S4</xref>); accounts that omit the contribution of these heuristics may thus lead to erroneous conclusions regarding uncertainty.</p>
</sec>
<sec id="s2f">
<label>3.6</label><title>Uncertainty has subject-specific effects on choices</title>
<sec id="s2f1">
<title>Heterogeneity in uncertainty effects across individuals</title>
<p>The model accounts for the potential effect of two aspects of uncertainty on choices. ΔEU captures either uncertainty-seeking (analogous to an uncertainty bonus <sup><xref ref-type="bibr" rid="c7">7</xref>,<xref ref-type="bibr" rid="c9">9</xref>,<xref ref-type="bibr" rid="c51">51</xref></sup>) or uncertainty avoidance, depending on the sign of the coefficient (w<sub>ΔEU</sub>). EUt captures a “default” choice strategy to repeat the previous choice or switch when overall uncertainty is higher (depending on the sign of the coefficient, w<sub>EUt</sub>). The coefficient estimates are on average close to zero, with a slight uncertainty avoidance effect (w<sub>ΔEU</sub> = −0.073, SE = 0.028, <italic>t</italic>(55) = −2.57, <italic>p</italic> = 0.01, Cohen’s <italic>d</italic> = 0.34) and a slight tendency to repeat the previous choice when overall uncertainty is higher (w<sub>EUt</sub> = 0.053, SE = 0.023, <italic>t</italic>(55) = 2.33, <italic>p</italic> = 0.02; Cohen’s <italic>d</italic> = 0.31. <xref rid="fig2" ref-type="fig">Fig. 2B</xref>; these parameter estimates are from the full model with interactions, see <xref rid="figs5" ref-type="fig">Fig. S5</xref> for the model without interactions). These results suggest that the heterogeneous contributions of these variables across individuals, which improve the model fit, manifest as null or small regression coefficients on average at the group level.</p>
</sec>
<sec id="s2f2">
<title>Uncertainty effects capture stable, psychologically meaningful individual differences</title>
<p>To test the hypothesis that the inter-subject variability of uncertainty-related parameters reflects true differences between individuals rather than noise in the data, we assess the stability of the coefficient estimates within subjects across time. We fit separate models to each subject’s data from the behavioral sessions on the one hand and fMRI sessions on the other hand, and compute the Spearman correlation of the resulting parameter estimates. The sessions were completed on different days, and with differences to adapt the task for the MRI scanner (see Methods). As a useful point of comparison, the coefficients of the strongest decision factors (repetition bias and ΔER) are highly correlated across sessions (repetition bias: <italic>r</italic> = 0.658, p = 10<sup>−7</sup>; ΔER: <italic>r</italic> = 0.618, <italic>p</italic> = 10<sup>−6</sup>; <xref rid="fig3" ref-type="fig">Fig. 3A</xref>, bottom), as are the coefficients for the win-stay-lose-shift heuristic terms: PE (r = 0.463, <italic>p</italic> = 10<sup>−4</sup>), UPE (r = 0.450, <italic>p</italic> = 10<sup>−3</sup>). Critically, the ΔEU (<italic>r</italic> = 0.49, <italic>p</italic> = 10<sup>−4</sup>) and, to a lesser extent, the EUt (<italic>r</italic> = 0.20, p = 0.13) effects are also stable across sessions. A qualitatively similar pattern is obtained for the full model with interactions, although in the fMRI session, the interaction terms are less reliably estimated than in the behavioral session, probably reflecting the two-fold difference in number of trials between sessions (<xref rid="figs6" ref-type="fig">Fig. S6</xref>). These results suggest that heterogeneity in the uncertainty effects reflects persistent individual differences rather than noise.</p>
<fig id="fig3" position="float" orientation="portrait" fig-type="figure">
<label>Figure 3.</label>
<caption><title>Multiple factors contribute to choices.</title>
<p><bold>A</bold>: Model selection of main effects. All considered decision factors contribute to choices. Adding 1-4 factors consecutively to a base model of repetition bias + ΔER improves the cross-validated model fit (average choice likelihood). Colors represent effect sizes (Cohen’s <italic>d</italic>) of model fit difference. Highlighted squares (in contrast to semi-transparent ones) are minimal model pairs with a one-factor difference, showing reliable improvements from adding each term. The bottom row and four rightmost columns (Ablation tests) show unique factor contributions controlling for all other factors; *: p&lt;0.05, **: p&lt;0.01, ***: p&lt;0.001 <bold>B.</bold> The selected (full) model coefficient estimates (all <italic>p</italic> &lt; 0.05; dots are individual subjects) show expected reward (ΔER; red) and repetition bias (blue) to be the dominant drivers of choices, with heterogeneous uncertainty effects (ΔEU and EUt; green) and an additional contribution of the signed and unsigned previous prediction error (PE and UPE, yellow), capturing a win-stay-lose-shift (WSLS) heuristic. Uncertainty interacts with the WSLS heuristic (light green bars), such that when relative uncertainty is higher on the previously unchosen option or total uncertainty is higher, subjects relied more on the heuristic.</p></caption>
<graphic xlink:href="587016v2_fig3.tif" mime-subtype="tiff" mimetype="image"/>
</fig>
<p>Next, we test whether these stable individual differences track with psychometric scales that measure psychological dimensions. Corroborating previous work showing that anxiety is associated with uncertainty avoidance in a bandit task <sup><xref ref-type="bibr" rid="c19">19</xref></sup>, individuals who score higher on anxiety are more likely to avoid the more uncertain option (ΔEU, estimated on all data; state anxiety: Pearson’s <italic>r</italic> = −0.32, <italic>p</italic> = 0.016; <xref rid="fig3" ref-type="fig">Fig. 3B</xref>; trait anxiety: <italic>r</italic> = −0.23, <italic>p</italic> = 0.08). A similar but weaker result is obtained for ΔEU coefficients using the full model with interactions (<xref rid="figs7" ref-type="fig">Fig. S7</xref>). Previous work has also found that compulsive gambling, associated with impulsivity, is related to stronger information-seeking in a bandit task <sup><xref ref-type="bibr" rid="c17">17</xref></sup>. In the present data, more impulsive individuals also tend to show greater uncertainty-seeking (Pearson’s <italic>r</italic> = 0.22, <italic>p</italic> = 0.09; <xref rid="fig3" ref-type="fig">Fig. 3B</xref>), in the full model including interactions (see <xref rid="figs7" ref-type="fig">Fig. S7</xref> for full correlation matrix of parameter estimates with psychometric scales).</p>
<p>Taken together, these results indicate that uncertainty plays a role in choices, with subject-specific, psychologically meaningful effects across individuals, which relate to trait measures related to anxiety and impulsivity.</p>
</sec>
</sec>
<sec id="s2g">
<label>3.7</label><title>Other decision heuristics drive choices</title>
<p>The individual differences in uncertainty effects we identify above can explain why previous studies have not always found a consistent uncertainty-seeking or uncertainty-avoiding tendency. In addition, some decision heuristics are partly confounded with uncertainty, and often not controlled for, which can also appear as uncertainty effects across studies. Specifically, in tasks with noisy observations, repeating the same choice typically reduces the uncertainty about the corresponding option, which could motivate the repetition. However, repetition may also simply reflect perseveration. This is a general behavioral tendency observed across many task contexts, which leads, in the context of reward-based choice, to selecting an option even when it is no longer the most rewarding one. Similarly, positively surprising observations – ones that generate large positive prediction errors – may result in repeating an option even when there is insufficient evidence that this option is most rewarding. Conversely, and perhaps more commonly, large negative prediction errors may cause switching to a less well-known option even when its expected reward is lower than the current option. When a win-stay-lose-shift strategy is not controlled for, these patterns may appear as uncertainty-seeking.</p>
<p>Consistently with these possibilities, we find that both repetition bias and terms that capture a win-stay-lose-shift strategy (the signed and unsigned previous prediction error, PE and UPE) significantly improve the model fit, after controlling for immediate reward and uncertainty-based effects (see <xref rid="s2e" ref-type="sec">Section 3.5</xref>). Furthermore, the coefficients for these effects show a consistent direction at the group level (while accounting for other factors): a bias to repeat rather than switch: w<sub>0</sub>= 1.087, SE= 0.111, <italic>t</italic>(55) = 9.76, <italic>p</italic> = 10<sup>−12</sup>, Cohen’s <italic>d</italic> = 1.30 and win-stay-lose-shift decisions: w<sub>PE</sub> = 0.412, SE = 0.069, <italic>t</italic>(55) = 6.02, <italic>p</italic> = 10<sup>−6</sup>, Cohen’s <italic>d</italic> = 0.80. A follow-up analysis of UPE shows that this term accounts for an asymmetry in win-stay-lose-shift decisions: subjects are more likely to repeat given large positive prediction errors than to switch in response to similarly large negative prediction errors (<xref rid="figs8" ref-type="fig">Fig. S8</xref>). Interestingly, the heuristic effects are also individually stable across sessions (<xref rid="fig4" ref-type="fig">Fig. 4A</xref>, bottom) and exhibit meaningful correlations with psychometric scales (e.g. between w<sub>PE</sub> and optimism; between w<sub>0</sub> and anxiety and impulsivity, see <xref rid="figs7" ref-type="fig">Fig. S7</xref>).</p>
<fig id="fig4" position="float" orientation="portrait" fig-type="figure">
<label>Figure 4.</label>
<caption><title>Effects of uncertainty on choices are subject-specific.</title>
<p><bold>A. Top</bold>. Scatterplots of regression coefficients estimated per session (behavioral and fMRI) show stability of uncertainty estimates. Dots are subjects. <bold>Bottom</bold>. Other model factors are also correlated across sessions. <bold>B.</bold> Uncertainty effects (ΔEU) correlate with psychological measures of anxiety and impulsivity. *p &lt; 0.05; ^0.05 &lt; p &lt; 0.1.</p></caption>
<graphic xlink:href="587016v2_fig4.tif" mime-subtype="tiff" mimetype="image"/>
</fig>
<p>In sum, we find that basic decision heuristics, which are often not controlled for in studies of exploration, consistently contribute to choices across subjects.</p>
</sec>
<sec id="s2h">
<label>3.8</label><title>Uncertainty modulates the use of the other heuristics</title>
<p>Heuristics are decision rules that subjects may be more likely to use when evidence about value is scarce or similar across options. In other words, uncertainty may modulate the extent to which behavior depends on heuristic decisions. To test this hypothesis, we first perform a two-step model selection analogous to that for the main effects model, but focused on testing interactions with uncertainty.</p>
<p>First, starting from a base model including all main effects, we add two-way interactions of each of the two uncertainty terms, ΔEU and EUt with either ΔER, or PE and UPE (together because they jointly capture the win-stay-lose-shift heuristic) (<xref rid="fig3" ref-type="fig">Fig. 3B</xref> and <xref rid="figs8" ref-type="fig">Fig. S8</xref>).</p>
<p>The result of this first step is that neither of the interactions with ΔER (ΔER*ΔEU and ΔER*EUt) individually contribute significantly to the model fit, they are therefore excluded (p<sub>cv</sub>(choice) with ΔER*ΔEU: <italic>t</italic>(55) = −0.20, <italic>n.s.</italic>; p<sub>cv</sub>(choice) with ΔER*EUt: <italic>t</italic>(55) = 0.65, <italic>n.s.</italic>). Both of the interactions of the uncertainties with the win-stay-lose-shift heuristic individually improve the model fit, so they are included in the second step. In the second step, we test whether adding both of these (pairs of) interactions further improves the model fit over either one in isolation. It does, so both are kept in the final model. Again, the fit improvement in terms of additional variance explained over the main-effects model is modest but reliable, with moderate effect size (0.17%, SE = 0.04, <italic>t</italic>(55) = 4.41, <italic>p</italic> = 10<sup>−4</sup>, Cohen’s <italic>d</italic> = 0.59); the fit improvement over the base model (intercept and ΔER) is comparable to that of adding the additional main effects (1.10%, SE = 0.18, <italic>t</italic>(55) = 6.29, <italic>p</italic> = 10<sup>−7</sup>, Cohen’s <italic>d</italic> = 0.84). There is therefore sufficient evidence that these interactions should be included in the model (<xref rid="figs9" ref-type="fig">Fig. S9</xref>).</p>
<p>To characterize their contributions, we examine the coefficient estimates for the interactions (<xref rid="fig3" ref-type="fig">Fig. 3B</xref>). When EUt is higher, subjects tend to rely more on the win-stay-lose shift heuristic. When the relative uncertainty, ΔEU, of the previously unchosen (switch) option is higher, they also rely more on this heuristic. Recall that the main effect of EUt, presented above, captures a tendency to repeat more when overall uncertainty is higher, which can also be seen as reinforcing a basic heuristic repetition bias. In sum, when faced either with more uncertainty across both options or more uncertainty in the unchosen option, subjects are more likely to base their choice on heuristics.</p>
</sec>
</sec>
<sec id="s3">
<title>Discussion</title>
<p>In this study, we used a novel bandit task to simultaneously characterize the contributions of uncertainty and other factors in reward-based decisions. We found a dual role for uncertainty in choices. First, uncertainty had subject-specific effects on choices, which were stable over time (correlated across testing sessions) and psychologically meaningful (correlated with psychometric measures of anxiety and impulsivity). Second, uncertainty modulated the use of two basic heuristics, repetition bias and a win-stay-lose-shift strategy: subjects rely more on these heuristics when overall uncertainty is higher and also when the other option (switch) is more uncertain than the current option (stay). Finally, the heuristics were themselves consistent drivers of choices: controlling for expected reward and uncertainty, subjects showed a general tendency to repeat rather than switch, and a tendency to overweigh the most recent observation, to repeat the previous choice in response to positive prediction errors and to switch in response to negative ones. However, it is important to note that, since the uncertainty-related coefficients were estimated in the presence of these heuristic terms, the subject-specific uncertainty-seeking/avoiding behavior is genuine.</p>
<p>These results provide a richer characterization of the interplay between value, uncertainty, and basic heuristic factors, where the latter are often neglected in studies of exploratory decision-making. They also inform an important puzzle in computational research on the exploration-exploitation dilemma: if uncertainty-based exploration strategies are generally an effective solution to explore-exploit tradeoffs, why are uncertainty effects not more robust and consistent across studies? The stable subject-specific uncertainty effects we found, ranging from uncertainty seeking to indifference and avoidance, point toward one part of an answer: that mixed results, especially from studies with smaller and non-random samples of subjects, may arise from underlying population-level heterogeneity in the uses of uncertainty. The heuristic decision factors we identify suggest another part of an answer. First, when not explicitly modeled, heuristic choices may appear either as exploratory effects or as uncertainty avoidance, depending both on the specific operational definition of exploration and on specifics of experimental design. Second, how individuals manage the balance between a plurality of decision factors in addition to reward and uncertainty may be a further source of heterogeneity. This balance is plausibly itself negotiated by recourse to uncertainty, as suggested by the uncertainty-by-heuristic interactions we report and by prior work <sup><xref ref-type="bibr" rid="c52">52</xref>,<xref ref-type="bibr" rid="c53">53</xref></sup>.</p>
<p>These interpretations, elaborated below, recast the question of variable uncertainty effects at the individual level: if uncertainty-based exploration is essential to effective reward-based decisions, why do individuals vary in whether and how they use uncertainty? There are two compatible possibilities. First, the correlations of uncertainty with psychometric measures of impulsivity and anxiety suggest individual differences in how well people leverage uncertainty in making decisions, which in extreme cases may be related to psychopathologies. Second, there may actually be a normative, resource-rational basis <sup><xref ref-type="bibr" rid="c39">39</xref>,<xref ref-type="bibr" rid="c40">40</xref></sup> for the individual differences observed in our study and elsewhere.</p>
<p>A growing number of studies have related individual differences in responses to uncertainty and exploratory behaviors to psychometric scores or clinical diagnoses, particularly of anxiety and impulsivity, consistent with our findings. With regard to impulsivity, a recent large-sample study of the general population has shown increased random exploration in more impulsive individuals, but no specific relationship to uncertainty-based exploration <sup><xref ref-type="bibr" rid="c18">18</xref></sup>. Two studies of patients with gambling disorder <sup><xref ref-type="bibr" rid="c17">17</xref></sup> and attention deficit hyperactivity disorder <sup><xref ref-type="bibr" rid="c54">54</xref></sup>, which are thought to involve impairments in impulse control, have respectively found increased information seeking and random exploration. Although these studies do not speak to the direct role of uncertainty in choices, they are broadly consistent with the present results in suggesting increased exploratory tendencies in more impulsive individuals. Interestingly, another study of gambling disorder <sup><xref ref-type="bibr" rid="c55">55</xref></sup>, found decreased uncertainty-directed exploration in gamblers, contrary to our results. This study used continuously drifting latent values<sup><xref ref-type="bibr" rid="c16">16</xref></sup>, rather than changepoints; this difference may be significant given that impulsivity might be specifically related to increased novelty-seeking <sup><xref ref-type="bibr" rid="c56">56</xref>,<xref ref-type="bibr" rid="c57">57</xref></sup> (cf. ref <sup><xref ref-type="bibr" rid="c17">17</xref></sup>). Over the long trial sequences in both tasks, none of the options is truly novel, but abrupt shifts in reward levels in our task might effectively induce a novelty-like effect.</p>
<p>With regard to anxiety, perhaps the strongest evidence for a negative relationship between anxiety scores and uncertainty-directed choices in line with our results comes from a recent study by Fan <italic>et al</italic> with large online samples, which found reduced uncertainty-directed exploration (as well as uncertainty avoidance) in a bandit task<sup><xref ref-type="bibr" rid="c19">19</xref></sup> (Fan et al., 2022). Notably, estimation uncertainty in the study by Fan <italic>et al</italic>, unlike ours, includes a component of outcome uncertainty differences across options, i.e., risk, leaving open the question of whether anxiety is related to risk aversion, as also suggested by another study <sup><xref ref-type="bibr" rid="c58">58</xref></sup>, or estimation uncertainty. Our study adds to these results that more anxious individuals tend to avoid uncertain (i.e. lesser known) options even when risk levels are matched.</p>
<p>Another study found that in a non-instrumental context, when uncertainty reduction does not bear on choosing a more rewarding option, more anxious individuals were willing to pay a premium to resolve uncertainty <sup><xref ref-type="bibr" rid="c59">59</xref></sup>, which in a bandit setting might suggest uncertainty seeking rather than avoidance. A general aversion to uncertainty characteristic of anxiety <sup><xref ref-type="bibr" rid="c60">60</xref></sup>, is consistent with either action to reduce uncertainty or with uncertainty avoidance, and the difference in whether uncertainty is instrumental or intrinsically valued is plausibly significant <sup><xref ref-type="bibr" rid="c21">21</xref></sup>. One recent study has, however, reported increased instrumental uncertainty-directed exploration with anxiety, contrary to our results <sup><xref ref-type="bibr" rid="c61">61</xref></sup>. The interpretation of this other study is complicated by the use of a proxy for estimation uncertainty that is potentially confounded with perseveration and choice stochasticity. In sum, although results are somewhat mixed, there are converging indications that exploratory choices are affected by anxious and impulsive traits, and a next step is to determine the precise nature of these effects, their relationship to uncertainty, and their causal origins.</p>
<p>We now turn to the other major conclusion of our study: choices are underpinned by multiple decision factors, including heuristic factors besides expected reward and uncertainty. The basic heuristic factors we identified, repetition bias and win-stay-lose-shift strategy, are naturally interpreted in terms of decision-making mechanisms that are commonly contrasted with reward-based choice, namely habitual and Pavlovian responses, respectively. Here, we provide an interpretation of these heuristics in these terms, while acknowledging the possibility of alternative interpretations.</p>
<p>The repetition bias may reflect a form of perseverative or habitual response, whereby the previous selection of an action increases the probability of selecting that action again. The probability of repeating may increase either because of a caching of previously rewarded stimulus-action associations <sup><xref ref-type="bibr" rid="c52">52</xref>,<xref ref-type="bibr" rid="c62">62</xref>,<xref ref-type="bibr" rid="c63">63</xref></sup>, or independently from reward <sup><xref ref-type="bibr" rid="c25">25</xref>–<xref ref-type="bibr" rid="c27">27</xref>,<xref ref-type="bibr" rid="c64">64</xref></sup>. In either case, the hallmark of habits is that they are fast and often efficient, but inflexible because of their decoupling from current rewards<sup><xref ref-type="bibr" rid="c65">65</xref></sup>. The caching mechanism is perhaps broadly analogous to the repetition bias in our study: once a subject has found the more rewarding option, they may choose it without referring to its expected value on every trial. However, computational studies of habit have recently emphasized that there is no necessary link between habits and values: actions may become habitual simply because they are repeated; it is incidental that the actions that tend to be repeated are usually rewarding ones <sup><xref ref-type="bibr" rid="c36">36</xref>,<xref ref-type="bibr" rid="c64">64</xref></sup>. This second view entails a more thorough dichotomy between two independent choice generating processes: a habitual and a reward-based one.</p>
<p>The win-stay-lose-shift heuristic we identified can be interpreted as an automatic tendency to approach stimuli previously paired with reward and to avoid ones paired with punishment (where “punishment” in the present context would be construed in relative terms, as lower-than-expected reward). This may occur even when such approach/avoidance behavior conflicts with the instrumentally dictated action, such as when expected reward recommends staying with an option that yielded surprisingly low reward on the last trial. This automatic tendency can be interpreted as “Pavlovian” in that it is governed by stimulus-outcome associations that are independent from instrumental considerations <sup><xref ref-type="bibr" rid="c29">29</xref>,<xref ref-type="bibr" rid="c66">66</xref>–<xref ref-type="bibr" rid="c68">68</xref></sup>. Innate Pavlovian tendencies to approach appetitive stimuli and avoid aversive ones can be hard to overcome, as shown in animal studies of associative learning <sup><xref ref-type="bibr" rid="c29">29</xref>,<xref ref-type="bibr" rid="c68">68</xref>–<xref ref-type="bibr" rid="c70">70</xref></sup>. In humans, Pavlovian biases have been demonstrated in Go/NoGo paradigms to conflict with instrumental requirements to withhold action to obtain a reward or to act to avoid punishment <sup><xref ref-type="bibr" rid="c32">32</xref>,<xref ref-type="bibr" rid="c66">66</xref>,<xref ref-type="bibr" rid="c67">67</xref></sup>. These studies argue for the independence of a Pavlovian choice generator, which, similar to habits, produces automatic responses to stimuli that are not coupled with action-outcome contingencies.</p>
<p>Thus, given the ubiquity of these basic heuristics in animal and human sequential decision-making, they can be plausibly construed as independent choice generators alongside more complex reward- and uncertainty-based strategies. Notably, however, alternative interpretations cannot be excluded. For example, it is possible that the apparent multiplicity in choice generative processes may be an artifact of model mis-specification: an inaccurate formalization of the value function or the learning process with respect to the true latent values guiding subjects’ choices may manifest as repetition bias or win-stay-lose-shift choices. Specifically, because the intercept term in the logistic regression (i.e., repetition bias) captures the mean tendency to repeat rather than switch, there is always the possibility that some residual repetitions are reward-driven from the point of view of a subject’s value estimation <sup><xref ref-type="bibr" rid="c71">71</xref>,<xref ref-type="bibr" rid="c72">72</xref></sup>, but not according to the Bayesian ideal observer. Similarly, with regard to win-stay-lose-shift, higher-than-optimal discounting of past rewards, due to, say, attentional and working memory limits, may manifest as choices that overweigh the last observation <sup><xref ref-type="bibr" rid="c23">23</xref>,<xref ref-type="bibr" rid="c73">73</xref></sup>.</p>
<p>Studies of the exploration-exploitation dilemma often focus on reward- and uncertainty-based drivers of choice without accounting for potential additional factors, such as perseverative or approach-avoidance responses. The present results call for consideration of the status of the basic heuristics with regard to exploration. On the broadest sense of exploration as any choices that do not maximize immediate reward <sup><xref ref-type="bibr" rid="c16">16</xref>,<xref ref-type="bibr" rid="c74">74</xref></sup>, basic heuristic-driven choices that diverge from immediate reward maximization are exploratory. It is not clear to what extent similar heuristic factors are present across the gamut of bandit settings used in the experimental literature. Stationary or slowly changing latent rewards would seem liable to result in perseverative, repetition effects. Perceptions of low controllability resulting from high reward stochasticity or volatility seem likely to increase the incidence of Pavlovian win-stay-lose-shift responses <sup><xref ref-type="bibr" rid="c66">66</xref></sup>.</p>
<p>Similarly, uncertainty-based definitions of exploration may confound a genuine effect of uncertainty with more basic heuristic factors, when these heuristics are not explicitly modeled. Such confounding may, on balance, go in either direction, mistakenly suggesting either uncertainty-seeking or uncertainty-avoidance, or either increases or decreases in choice stochasticity with uncertainty. There are many conceivable patterns of confounding, and they may help explain the heterogeneity of uncertainty-based exploration effects in the literature. Future work should systematically characterize the recruitment of different decision heuristics with variations in task parameters. When they are not directly of interest, they should at least be controlled for.</p>
<p>Lastly, the interactions we identify of uncertainty with the basic heuristics bear on the preceding discussion in several ways. Methodologically, they may further aggravate confounding when relevant decision factors are unaccounted for. Substantively, two notable implications of these interactions are that, first, they may point toward an uncertainty-based arbitration mechanism between strategies <sup><xref ref-type="bibr" rid="c52">52</xref>,<xref ref-type="bibr" rid="c53">53</xref></sup>, and second, individual differences in the strength and direction of such interactions may be a further source of heterogeneity in choice patterns. A promising direction for evaluating these possibilities in future work is to move away from models of average factor contributions over entire choice sequences and toward modeling the dynamics of strategy selection with temporally resolved methods, both descriptively (for e.g., with HMM-GLMs <sup><xref ref-type="bibr" rid="c75">75</xref></sup>) and algorithmically.</p>
<p>In conclusion, the present study has showcased the rich structure of human sequential reward-based decisions, and has identified stable and meaningful individual differences in the uses of uncertainty in choice, which can account for mixed results of prior work. These results underscore the importance of characterizing the interplay between reward, uncertainty, and a broader range of cognitively relevant factors for understanding the processes governing human choice under uncertainty, their functional basis and dysfunction in psychopathology.</p>
</sec>
<sec id="s4">
<title>Methods</title>
<sec id="s4a">
<label>2.1.</label><title>Subjects</title>
<p>Sixty-two adult volunteers with no self-reported history of psychiatric or neurological conditions participated in the study and received a fixed monetary compensation. Two subjects were excluded for only completing one of the two sessions of the experiment. One subject was excluded for an incidental finding of brain anomaly in the MRI. Three subjects were excluded for chance-level task performance. Chance level (mean and SD) was defined as the average obtained reward across 1000 iterations of random choices on the same reward sequences as subjects, and the exclusion threshold was defined as mean obtained rewards less than two standard deviations above chance: M<sub>chance_reward</sub> = 49.5 points, SD<sub>chance_reward</sub> = 0.56, cutoff =50.6; the 3 excluded subjects obtained 50.1, 49.0, and 49.7 points on average. The data from the remaining 56 subjects were analyzed (M<sub>age</sub> = 25.9 years, SD<sub>age</sub> = 6.5, range 18-44, 29 women, 27 men; 48 self-reported right-handed; all native French speakers). The study was approved by a national ethics review board (Comité de protection des personnes Ile de France III, approval #2018-A09195-50), and subjects gave informed consent prior to participating.</p>
</sec>
<sec id="s4b">
<label>2.2.</label><title>Main task and procedure</title>
<sec id="s4b1">
<title>Sessions</title>
<p>Subjects completed two 2-hour long in-lab sessions: first, a behavioral session composed of 8 blocks of the task (as well as instructions and practice, see below), and an MRI session on a separate day, consisting of 4 blocks. The sessions were between 1 and 29 days apart (M<sub>days</sub>=10, SD<sub>days</sub>=6.5). Subjects were encouraged to take occasional breaks in-between blocks to maintain attentiveness. Between the two in-lab sessions, subjects completed an online battery of psychometric questionnaires, on the Gorilla Experiment Platform (<ext-link ext-link-type="uri" xlink:href="https://gorilla.sc/">https://gorilla.sc/</ext-link>), detailed below.</p>
</sec>
<sec id="s4b2">
<title>Task structure</title>
<p>We developed a novel version of a non-stationary two-armed bandit task, which allowed (a) dissociating reward-driven choices from other decision factors, and (b) better identifying uncertainty-based vs. heuristic-driven choices. A block of the task (<xref rid="fig1" ref-type="fig">Fig. 1A, C</xref>) consisted of 96 partial-feedback trials (i.e., subjects only observed the outcome for the chosen option, creating an explore-exploit tradeoff), with rewards ranging between 1 and 100 points drawn from Gaussians centered around 3 mean reward levels of 30, 50, and 70 points. There were uncued, independent changes of the reward levels for each option every 24 trials, on average (min: 9, max: 36), for a total of 4 changepoints per option per block (volatility, i.e. probability of a change point on a given option at any given trial, = 1/24; with a minimum of 9 trials in-between consecutive changes on the same option). There were interleaved segments of 4 forced-choice and 8 free-choice trials (each block starting with a forced choice segment), whose goal was to decorrelate expected reward and estimation uncertainty, which are otherwise anticorrelated (i.e., the more rewarding option is selected more often thereby reducing its estimation uncertainty; <xref rid="figs10" ref-type="fig">Fig. S10</xref>) <sup><xref ref-type="bibr" rid="c11">11</xref></sup>. Further, each block consisted of a cued high-noise (SD=20 points) and a low-noise (SD=10 points) period of 48 consecutive trials (in counterbalanced order across runs). Notably, the level of noise for the two options was matched throughout, therefore the relative risk level or irreducible uncertainty was always matched between the two options. The goal of this noise-level manipulation was to induce different levels of estimation uncertainty across trials, on top of the differences induced by change points. Within a block, the average rewards and variability for each option were matched.</p>
</sec>
<sec id="s4b3">
<title>Reward sequences</title>
<p>A total of 8 fixed sequences were used to maintain comparability across subjects and sessions, for purposes of investigating individual differences in model parameter estimates across subjects as well as parameter stability within subjects across sessions. Even though the underlying rewards on the two options were “fixed”, the exact sequences of rewards observed by each subject and across sessions varied for the free-choice periods because of partial feedback. The forced choices were fixed for each sequence (i.e., all subjects were required to choose the same option on a given trial, and consequently they obtained the same reward), and were balanced for equal (each option sampled twice in a 4 trial segment) vs unequal (options sampled 1 and 3 times) information across options <sup><xref ref-type="bibr" rid="c7">7</xref></sup>. In the behavioral session, subjects completed 8 blocks, with each block consisting of a single presentation of each sequence in randomized order. In the fMRI session, subjects completed a subset of 4 of the same 8 sequences.</p>
</sec>
<sec id="s4b4">
<title>Visual presentation</title>
<p>The two arms of the bandit, referred to as “options”, were presented as a filled purple and orange circle on a gray background, and outcomes for the chosen option were shown in black numerals inside the respective circle. The location of the purple and orange option (left vs right) was counterbalanced across blocks but was held constant within a block. Notably, in this design, a tendency to repeat the previous option irrespective of reward history conflates motor perseveration (making the same motor response) and choice perseveration (choosing the same option).</p>
<p>A second, larger concentric circle around the option signaled free- vs. forced-choice trials: when only one of the options was “outlined” with a second circle, subjects had to select that option; when both options were highlighted, subjects could freely choose an option. Upon option selection, the circle around the selected option disappeared to provide feedback that the response was registered, and then reappeared for 500 ms before the obtained reward was displayed. Fixation was indicated by a bullseye at the center of the screen (lighter-gray relative to the background), which also served to signal whether subjects were currently in a low- or high-noise period: 2 concentric circles indicated low noise; 3 circles indicated high noise. When the noise level switched halfway through a block, a yellow circle flashed around the fixation bullseye to alert subjects to the change.</p>
<p>There was no indication of reward history from past trials, the number of trials elapsed and remaining in a block.</p>
<p>Subjects were instructed to maintain central fixation, and the visual display of the options subtended roughly 2.5 degrees of visual angle (from the outer edges of the left and right options). Subjects wore noise-canceling headphones in both sessions (Behavioral session: Bose noise canceling headphones 700; <ext-link ext-link-type="uri" xlink:href="https://www.bose.com/">https://www.bose.com/</ext-link>; MRI session: MR Confon, <ext-link ext-link-type="uri" xlink:href="https://www.crsltd.com/tools-for-functional-imaging/audio-for-fmri/mr-confon/">https://www.crsltd.com/tools-for-functional-imaging/audio-for-fmri/mr-confon/</ext-link>). In the behavioral session white noise was played over the headphones.</p>
</sec>
<sec id="s4b5">
<title>Trial timing</title>
<p>The timing differed across the behavioral and MRI sessions. Specifically, the behavioral session was quasi-self-paced, with the following fixed components: a 1 s interval between button press and outcome, a 2 s presentation of the outcome, and a 1 s inter-trial interval (ITI) consisting of a fixation-only display between the disappearance of the last trial’s outcome and the reappearance of the cues for the next trial.. By contrast, in the fMRI session, the entire trial duration was fixed as follows: 2 s cue display and response window, 1 s interval between cue and outcome display, 2 s outcome display, 2 to 5 s ITI (3.5 s on average). Therefore, the average stimulus onset asynchrony in the scanner was 8.5 s (range 7-10 s). If subjects failed to respond within the 2 s response window, a red fixation cross was displayed for the remaining duration of the trial (M=5.02 (1.07%), SD=6.36).</p>
</sec>
<sec id="s4b6">
<title>Responses</title>
<p>In the behavioral session, subjects indicated their choices on a standard QWERTY keyboard, with an index-finger button press of the left and right hand for the left and right options, respectively, using the “f” and “j” keys. In the scanner, subjects made their responses with an index- or middle-finger button press (left and right options, respectively), via a custom-made MR-compatible optic fiber button box with 5 buttons, developed in-house.</p>
</sec>
<sec id="s4b7">
<title>Explicit reports</title>
<p>Subjects were occasionally asked to report guesses for the reward level of each option (30, 50, or 70 points) and their confidence in these guesses (4-point Likert scale from 0 = “Not at all” to 3 = “Totally”, corresponding to “Pas du tout” and “Totalement” in French). A “Questions” screen signaled the beginning of an explicit report period. Subjects first answered both questions for one option and then for the other.. In the behavioral session, subjects indicated value guesses on a keyboard with their left hand (keys: “s”=30, “d”=50, “f”=70) and confidence with their right hand (“j”=0, “;”=3). In the fMRI session, they answered both questions with the right hand, with thumb, index, and middle fingers for value guesses and additionally ring finger for confidence (low to high reward level / confidence mapping from thumb through middle / ring finger). The order in which the options were queried was counterbalanced within a block.</p>
<p>In each behavioral block, there were a total of 6 reports (each one querying both options), 4 of which were on free trials and 2 on forced trials. In an fMRI block, there were 4 reports, 2 on free and 2 on forced trials. The total number of explicit reports across both sessions was 128.</p>
<p>The timing of the responses was self-paced in the behavioral session and with a 3 s response window in the fMRI session. In the fMRI session, the next question appeared once a response was registered, but the total duration of a report period was kept fixed by adjusting the duration of fixation display following the last report.</p>
</sec>
<sec id="s4b8">
<title>Instructions and practice</title>
<p>In the first in-lab session, prior to the main task, subjects received written task instructions in French that relied on a backstory to facilitate understanding of the task structure. Subjects were given an opportunity to ask clarification questions. The backstory involved commodity traders of apples and oranges in noisy and volatile market conditions (see <xref rid="figs11" ref-type="fig">Fig. S11</xref> for the English translation of the instruction with the backstory). They then completed two practice blocks: first, a full block (96 trials) of an illustrated version of the task matching a backstory; and second, a short practice block of the task as presented with a simplified display during the main experiment (12 trials; including 2 sets explicit reports for familiarization). The instructions included, through the backstory, complete information of the task structure to minimize effects of structure learning during the early trials / blocks. Specifically, subjects were instructed of (a) the 3 reward levels, volatility (changes every 24 trials on average), and the fact that reward levels would change independently across optionarms and independently of the subjects’ responses, without accompanying cues (b) the noisiness of the obtained rewards about the mean reward levels, as well as the two levels of noise, (c) the presence of interleaved forced choice trials, and (d) the occasional explicit reports. Subjects were also instructed that their goal is to maximize rewards (total number of points obtained), although their actual compensation was not dependent on task performance. Following the full-block practice, subjects were asked a series of questions to ensure task comprehension, and verbal clarification was provided as needed.</p>
<p>In the second session, prior to going inside the MRI scanner, subjects completed another short practice block (12 trials) on a laptop computer, as a reminder of the task structure, as well as to familiarize them with the trial timing differences from the behavioral session, and the different response mappings in the scanner (one-hand in the scanner vs. both hands in the behavioral session).</p>
</sec>
</sec>
<sec id="s4c">
<label>2.3.</label><title>Psychometric scales</title>
<p>Between the two sessions, at a time of their choosing, subjects completed an online battery of psychometric scales comprising previously validated French translations of the following 8 questionnaires:
<list list-type="bullet">
<list-item><p>Weiss Functional Impairment Rating Scale (WFRIS) for Attention Deficit and Hyperactivity Disorder (ADHD), with subscales for different domains including family, work, school, life skills, self-concept, social, and risk-taking <sup><xref ref-type="bibr" rid="c76">76</xref></sup>; French translation <sup><xref ref-type="bibr" rid="c77">77</xref></sup></p></list-item>
<list-item><p>State-trait anxiety inventory, form Y (STAI-Y) for clinical anxiety, with subscales for state and trait anxiety <sup><xref ref-type="bibr" rid="c68">68</xref></sup>; French translation <sup><xref ref-type="bibr" rid="c78">78</xref></sup></p></list-item>
<list-item><p>Revised Life Orientation Test (LOT-R) for optimism <sup><xref ref-type="bibr" rid="c79">79</xref></sup>; French translation <sup><xref ref-type="bibr" rid="c80">80</xref></sup>;</p></list-item>
<list-item><p>Barratt Impulsiveness Scale (BIS-11) for impulsivity <sup><xref ref-type="bibr" rid="c81">81</xref></sup>; French translation <sup><xref ref-type="bibr" rid="c82">82</xref></sup>;</p></list-item>
<list-item><p>Snaith-Hamilton Pleasure Scale (SHAPS) for anhedonia <sup><xref ref-type="bibr" rid="c83">83</xref></sup>; French translation <sup><xref ref-type="bibr" rid="c84">84</xref></sup>;</p></list-item>
<list-item><p>Autism quotient (AQ) for autism spectrum <sup><xref ref-type="bibr" rid="c85">85</xref></sup>; French translation <sup><xref ref-type="bibr" rid="c86">86</xref></sup></p></list-item>
<list-item><p>Schizotypal personality questionnaire (SPQ) for schizophrenic symptomatology <sup><xref ref-type="bibr" rid="c87">87</xref></sup>; French translation <sup><xref ref-type="bibr" rid="c88">88</xref></sup></p></list-item>
<list-item><p>Big Five Personality Inventory (BIG 5), measuring the “Big 5” personality traits: openness, conscientiousness, extraversion, agreeableness, and neuroticism <sup><xref ref-type="bibr" rid="c89">89</xref></sup>; French translation <sup><xref ref-type="bibr" rid="c90">90</xref></sup>.</p></list-item>
</list></p>
<p>The scales were presented in a fixed order, as listed above, across all subjects, on the Gorilla Experiment Platform (<ext-link ext-link-type="uri" xlink:href="https://gorilla.sc/">https://gorilla.sc/</ext-link>). The French translations of the questionnaires are available on github &lt;add link&gt;.</p>
</sec>
<sec id="s4d">
<label>2.4.</label><title>Models</title>
<sec id="s4d1">
<title>Learning model: Bayesian ideal observer</title>
<p>The Bayesian learner was specified in a Hidden Markov Model (HMM) framework. Starting from a uniform prior over the 3 reward levels for each option p(μ<sub>0</sub>), on each trial, the model updates p(μ<sup>(k)</sup><sub>i,t+1</sub>|y<sub>1:t+1</sub>, σ<sub>1:t+1</sub>, ν), the posterior probability of the hidden reward level μ<sub>i</sub> corresponding to option <italic>k</italic> using the new outcome y<sub>t+1</sub>, and assuming some fixed probability ν of a changepoint, specified via a transition matrix (with 1-p(changepoint) on the diagonal; p(changepoint) evenly distributed over the remaining columns of each row). The likelihood p(y<sub>t+1</sub>|μ<sup>(k)</sup><sub>i,t+1</sub>, σ<sub>t+1</sub>) evaluates the probability of the current outcome y<sub>t+1</sub> following the choice of option <italic>k</italic>, at each reward level μ<sub>i</sub>, using a normal density function (with a mean μ<sub>i</sub> and the standard deviation σ<sub>t+1</sub> corresponding to the generative noise level. Note that when no outcome is available for a given option, which occurs whenever this option was not selected, the likelihood function is constant (it does not depend on μ<sub>i, t+1</sub>). The update relies on Bayes rule and leverages the fact that, conditioned on the previous reward level μ<sup>(k)</sup><sub>i,t</sub>, the new reward level μ<sup>(k)</sup><sub>i,t+1</sub> is independent from previous outcomes y<sub>1:t</sub>, and the fact that, conditioned on the current reward level μ<sub>i,t+1</sub>, the current outcome y<sub>t+1</sub> is independent from the previous ones y<sub>1:t</sub>. The update equation for the posterior is:
<disp-formula>
<graphic xlink:href="587016v2_ueqn1.gif" mime-subtype="gif" mimetype="image"/>
</disp-formula></p>
<p>We used this posterior to model the maximum a posteriori reward level and estimation uncertainty at the moment of questions.</p>
<p>For choices, we modeled the expected reward level, estimation uncertainty and prediction error using the predictive distribution of reward levels (by updating the posterior from one trial to the next assuming that a change point could occur):
<disp-formula>
<graphic xlink:href="587016v2_ueqn2.gif" mime-subtype="gif" mimetype="image"/>
</disp-formula></p>
</sec>
<sec id="s4d2">
<title>Learning model validation</title>
<p>A key reason for using a Bayesian learner in this study is that it provides trial-by-trial estimates of uncertainty, which can be used to assess subjects’ uncertainty-driven behavior. To validate that this learning model adequately fits the reward estimates of subjects, we compared the Bayesian learning model against a Rescorla-Wagner (RW) delta-rule learning model <sup><xref ref-type="bibr" rid="c47">47</xref>,<xref ref-type="bibr" rid="c91">91</xref></sup>. The standard delta-rule learning model updates the reward estimate of the chosen option only. Given the poor performance of this model, we used a variant to also update the value of the unchosen option, making it decay toward the average reward level (50):
<disp-formula>
<graphic xlink:href="587016v2_ueqn3.gif" mime-subtype="gif" mimetype="image"/>
</disp-formula></p>
<p>Where α is the learning (and forgetting) rate, fit to the subjects choice (assuming as a choice model the baseline model presented in the Result section).</p>
</sec>
<sec id="s4d3">
<title>Model of choice stochasticity</title>
<p>We considered that choices (repeat or switch) are a stochastic function of a linear combination of decision factors <italic>f<sub>i</sub></italic>. We distinguished two forms of stochasticity. One assumed that choice probability increases following the linear combination of decision factors. More specifically, choice probability is a logistic function of the linear combination of factors (when there are two possible choices, this choice model is related to the softmax policy<sup><xref ref-type="bibr" rid="c47">47</xref></sup>):
<disp-formula>
<graphic xlink:href="587016v2_ueqn4.gif" mime-subtype="gif" mimetype="image"/>
</disp-formula></p>
<p>With <italic>l</italic> the logistic function: <inline-formula><inline-graphic xlink:href="587016v2_inline1.gif" mime-subtype="gif" mimetype="image"/></inline-formula></p>
<p>The other model of choice stochasticity, known as epsilon-greedy<sup><xref ref-type="bibr" rid="c47">47</xref></sup> assumed that subjects chose to repeat if <italic>w</italic><sub>0</sub> + ∑<italic><sub>i</sub> w<sub>i</sub>f<sub>i</sub></italic>(<italic>t</italic>) is positive (and to switch otherwise) with probability 1-ε (constant and independent across trials) or else chose the other option.</p>
</sec>
<sec id="s4d4">
<title>Decision factors</title>
<p>The decision factors were derived from the ideal observer posterior probabilities of the reward levels:
<list list-type="bullet">
<list-item><p><italic>Relative expected reward (ΔER)</italic>. It is the difference (repeat minus stay) in expected reward associated with each option. The expected reward for option <italic>k</italic> on trial <italic>t</italic> is computed from the predictive distribution of reward levels: <inline-formula><inline-graphic xlink:href="587016v2_inline2.gif" mime-subtype="gif" mimetype="image"/></inline-formula>.</p></list-item>
<list-item><p><italic>Estimation uncertainty (ΔEU</italic> and <italic>EUt)</italic>. We modeled the estimation uncertainty of the reward levels of option <italic>k</italic> on trial <italic>t</italic> as the residual probability mass that is not the maximum predictive probability across reward levels: <inline-formula><inline-graphic xlink:href="587016v2_inline3.gif" mime-subtype="gif" mimetype="image"/></inline-formula>. This quantity is particularly sensitive to the noise level and recent changepoints, as well as the option sampling history. Two aspects of estimation uncertainty were considered: relative (ΔEU; i.e., the difference between options) and total (EUt; i.e., the average across options; it captures a modulation of the repetition bias by total estimation uncertainty).</p></list-item>
<list-item><p><italic>Signed and unsigned prediction error on the previous trial (PE and UPE)</italic>. The prediction error is the difference between the expected and obtained reward on the previously chosen arm <inline-formula><inline-graphic xlink:href="587016v2_inline4.gif" mime-subtype="gif" mimetype="image"/></inline-formula>. When controlling for the other factors the effects of the signed and unsigned predictor error reflect the second heuristic of interest in the study: a “win-stay-lose-shift” strategy, whereby decisions on the current trial are based on the direction and amount of deviation of the outcome of the previous choice from expectation. The effect of the unsigned prediction error (|PE<sub>t-1</sub>|) captures a potential asymmetry in the slope of the PE effect between the positive and negative domains (see Results and <xref rid="figs8" ref-type="fig">Fig. S8</xref>).</p></list-item>
</list></p>
<p>We considered a baseline model consisting of only the constant w<sub>0</sub> and <italic>ΔER</italic>. We then considered more complex models by adding each factor and each combination of factors in a full factorial way.</p>
<p>Therefore, the baseline model was:
<disp-formula>
<graphic xlink:href="587016v2_ueqn5.gif" mime-subtype="gif" mimetype="image"/>
</disp-formula></p>
<p>And the most complex main-effects model was:
<disp-formula>
<graphic xlink:href="587016v2_ueqn6.gif" mime-subtype="gif" mimetype="image"/>
</disp-formula></p>
</sec>
<sec id="s4d5">
<title>Interactions</title>
<p>To examine potential interactions of uncertainty (ΔEU, EUt) with other factors (ΔER, PE, UPE), combinations of candidate two-way interactions were added as additional regressors to the winning main-effects-only model, and the best-fitting model was the final selected model, on which the main results focus.</p>
</sec>
<sec id="s4d6">
<title>Model fit</title>
<p>Regressors were standardized (z-scored) prior to model fitting for comparability of the resulting coefficient estimates. Model fitting was performed in Python, with scipy.optimize, using maximum-likelihood estimation (MLE) on the log-loss between model predictions and subjects’ choices. For models that fitted volatility on top of the decision parameters, the Powell optimization procedure was used. For models that used the generative volatility, only the decision parameters remained to be fitted; the decision model being equivalent to a logistic regression model, we fitted a logistic regression model using scipy’s defaults (the Broyden-Fletcher-Goldfarb-Shanno (BFGS) procedure, called using the wrapper statsmodels.logit was used as a backend).</p>
</sec>
<sec id="s4d7">
<title>Model comparison</title>
<p>Models were compared with leave-one-block-out cross-validation (i.e., 11 blocks training, 1 block testing). More precisely, we used the cross-validated average choice likelihood, that is, the average model-derived probability of subjects’ choices on each free trial. The average choice likelihood is a sensitive measure of model fit for comparison, because it is less impacted by occasional low-probably choices than the quantity being minimized, and it has an intuitive interpretation: it reflects the extent to which subjects’ choices are, on average, consistent with model predictions. For the main goal of comparing decision models with different subsets of candidate factors, cross-validation was performed on the decision-model only, using subject-specific volatility estimates derived on all data (rather than fitting on the training set only). This was done to reduce the amount of computing time needed to fit the models. As an additional check, after the final main-effects model was selected, single-factor ablations were performed from this final model with full cross-validation (including the fit of volatility on the training set only). The final model reliably outperformed all ablated models except for a marginal contribution of one factor, the total estimation uncertainty (Δp(choice) = 0.001, <italic>t</italic>(55) = 1.92, <italic>p</italic> = 0.06, Cohen’s <italic>d</italic> = 0.26) (see <xref rid="figs12" ref-type="fig">Fig. S12</xref> for full results).</p>
</sec>
<sec id="s4d8">
<title>Parameter and model recovery</title>
<p>Parameter recovery was performed to ensure that effects in the selected model are sufficiently dissociable to estimate independent coefficients. The parameter recovery procedure was as follows: (1) sample parameter values uniformly within the range of subjects’ actually observed parameter estimates, independently for each parameter; (2) simulate choices with these parameter settings; (3) fit a model to the simulated choices to obtain recovered parameter values; (4) iterate 1000 times; and (5) compute Spearman correlations of the generative and recovered parameter estimates across iterations, for all pairs of parameters, to produce a confusion matrix. (<xref rid="figs1" ref-type="fig">Fig. S1</xref>)</p>
<p>For the model recovery procedure, steps (1)-(2) were identical. For (3), all main-factor models (n=16) were fit to choices generated with a given model; for (4), there were 100 iterations. To evaluate the results, (5), we counted the number of times a given generative model was best fit by each model. (<xref rid="figs2" ref-type="fig">Fig. S2</xref>)</p>
<sec id="s4d8a">
<title>Statistical analysis</title>
<p>The group-level inferential statistics were as follows. We performed one-sample (against 0, for e.g., on logistic regression coefficients; or chance level, where different from 0; for e.g, on proportion of explicit reports matching the Bayesian learner) and paired-sample t-tests (for e.g., on cross-validated model fits for pairs of models). Reported correlation coefficients are either Pearson’s product-moment correlations or Spearman’s rank correlations, as noted in the text.</p>
</sec>
</sec>
</sec>
<sec id="s4e">
<label>2.5.</label><title>Code and data</title>
<p>Code and data are available here (see README.txt).</p>
<p>This copy of the code and data is for peer review purposes. Please do not share this link.</p>
</sec>
</sec>
</body>
<back>
<ack>
<title>Acknowledgements</title>
<p>This work was funded by CRCNS grant NIH/NIDA R01DA050373 to AY; ANR 19-NEUC-0002-01 to FM. We thank the nurses, radiographers and other staff at NeuroSpin, INSERM-CEA Cognitive Neuroimaging Unit for support with data collection.</p>
</ack>
<sec id="s5">
<title>Additional information</title>
<sec id="s5a">
<title>Contributions</title>
<p>FM, AY, DG, and AP conceived and designed the study; FM &amp; AY supervised the project; ML &amp; AP collected the data; FM &amp; DG contributed models, AP analyzed the data; AP, ZH, AY, &amp; FM prepared the study for publication. AP wrote the manuscript with discussion and comments by all co-authors.</p>
</sec>
<sec id="s6">
<title>Conflicts of interest</title>
<p>None to declare.</p>
</sec>
</sec>
<sec id="s7">
<title>Supplementary figures</title>
<fig id="figs1" position="float" orientation="portrait" fig-type="figure">
<label>Figure S1:</label>
<caption><title>Parameter recovery.</title>
<p>Spearman’s correlations between generative and recovered parameter estimates across 1000 iterations of simulated choices for the main effects model (left) and the full model, including interactions (right). *p &lt; 0.05, **p &lt; 0.01, ***p &lt; 0.001.</p></caption>
<graphic xlink:href="587016v2_figs1.tif" mime-subtype="tiff" mimetype="image"/>
</fig>
<fig id="figs2" position="float" orientation="portrait" fig-type="figure">
<label>Figure S2.</label>
<caption><title>Model recovery.</title>
<p>Confusion matrix of proportion of simulated choice sequences (out of 100 iterations) generated with a given model (x-axis) that are best fitted by each model (y-axis) (including only main effects models; n=16). Note that the larger proportion of matches in the lower-left triangle compared to the upper-right triangle of the matrix indicates that behavior generated by a given model can be well accounted for by a more complex model, but not by a simpler model. In particular, behavior generated by the most complex model with all main effects (which best accounted for our subjects’ data) is not well fitted by any of the simpler models.</p></caption>
<graphic xlink:href="587016v2_figs2.tif" mime-subtype="tiff" mimetype="image"/>
</fig>
<fig id="figs3" position="float" orientation="portrait" fig-type="figure">
<label>Figure S3.</label>
<caption><title>Counts of best fitting model across subjects.</title>
<p>Number of subjects who are best fitted by each of the main effects models. Note that the full model and the models excluding only either ΔEU or EUt are the best-fitting models for 24 of 56 subjects (43%), and models that include one or more uncertainty term best-fitted 44 of 56 subjects (79%).</p></caption>
<graphic xlink:href="587016v2_figs3.tif" mime-subtype="tiff" mimetype="image"/>
</fig>
<fig id="figs4" position="float" orientation="portrait" fig-type="figure">
<label>Figure S4.</label>
<caption><title>ΔEU parameter misestimation without heuristics.</title>
<p>Omitting either the repetition bias (“no rep. bias”) or both repetition bias and win-stay-lose-shift heuristic terms (PE and UPE) (“no heuristics”) results in significantly different group-level ΔEU effects (see also <xref rid="figs5" ref-type="fig">Fig. S5</xref>). This suggests ΔEU can be misestimated if the heuristics are not accounted for.</p></caption>
<graphic xlink:href="587016v2_figs4.tif" mime-subtype="tiff" mimetype="image"/>
</fig>
<fig id="figs5" position="float" orientation="portrait" fig-type="figure">
<label>Figure S5.</label>
<caption><title>Main effects model.</title>
<p>Parameter estimates for the model with main effects only (no interactions). * indicates p &lt; 0.05.</p></caption>
<graphic xlink:href="587016v2_figs5.tif" mime-subtype="tiff" mimetype="image"/>
</fig>
<fig id="figs6" position="float" orientation="portrait" fig-type="figure">
<label>Figure S6.</label>
<caption><title>Parameter stability of the model including interactions.</title>
<p>Spearman’s rank correlations of parameter estimates across testing sessions for full model, including interactions. *p &lt; 0.05; ^p &lt; 0.1.</p></caption>
<graphic xlink:href="587016v2_figs6.tif" mime-subtype="tiff" mimetype="image"/>
</fig>
<fig id="figs7" position="float" orientation="portrait" fig-type="figure">
<label>Figure S7:</label>
<caption><title>Psychometric scales correlations with parameter estimates from the models.</title>
<p>Pearson correlations of all included psychometric scales and subscales with <bold>A:</bold> parameters of main effects-only model and <bold>B:</bold> parameters of full model including interactions. ^p &lt; 0.10; *p &lt; 0.05, **p &lt; 0.01. Scale acronyms: <italic>WFRIS</italic>: Weiss Functional Impairment Rating Scale; <italic>STAI-Y</italic>: State-Trait Anxiety Inventory; <italic>LOT-R</italic>: Life Orientation Test - Revised; <italic>BIS-11</italic>: Barratt Impulsiveness scale. SHAPS: Snaith–Hamilton Pleasure Scale; AQ: Autism quotient; SPQ: Schizotypal personality questionnaire; BIG 5: Big five personality questionnaire (extroversion, agreeableness, conscientiousness, neuroticism, openness to experience).</p></caption>
<graphic xlink:href="587016v2_figs7.tif" mime-subtype="tiff" mimetype="image"/>
</fig>
<fig id="figs8" position="float" orientation="portrait" fig-type="figure">
<label>Figure S8.</label>
<caption><title>Asymmetry of prediction error effect.</title>
<p>Test of the interpretation of the unsigned prediction error term (UPE) as capturing an asymmetry effect, as used in the model presented in the main text. The previous prediction error (PE, capturing win-stay-lose-shift heuristic) in the main-effects model is parameterized into “sign” (indicator 0/1 variable), “magnitude” (i.e., unsigned prediction error), and interaction terms (sign x magnitude). The interaction term captures the asymmetry of the PE effect in the positive and negative domains (beta = 0.137, SE = 0.034, <italic>t</italic>(55) = 4.037, <italic>p</italic> = 0.0002, <italic>Cohen’s d</italic> = 0.54). *p &lt; 0.05. In the main text, this asymmetry is captured by the UPE (unsigned prediction error). The advantage of a model with PE and UPE over PE sign, PE magnitude and their interactions is model simplicity and easier modeling of interaction with uncertainty.</p></caption>
<graphic xlink:href="587016v2_figs8.tif" mime-subtype="tiff" mimetype="image"/>
</fig>
<fig id="figs9" position="float" orientation="portrait" fig-type="figure">
<label>Figure S9.</label>
<caption><title>Model selection with interactions.</title>
<p>Model fit improvements from adding interaction terms to the selected main effects model. The mode complex model is better than any simpler model. Model numbers correspond to:</p><p>1. repeat ∼ 1+ΔER+ΔEU+EUt+PE+UPE (base model without interactions)</p><p>2. repeat ∼ 1+ΔER+ΔEU+EUt+PE+UPE<bold>+ΔER*ΔEU</bold></p><p>3. repeat ∼ 1+ΔER+ΔEU+EUt+PE+UPE<bold>+ΔER*EUt</bold></p><p>4. repeat ∼ 1+ΔER+ΔEU+EUt+PE+UPE<bold>+PE*ΔEU+UPE*ΔEU</bold></p><p>5. repeat ∼ 1+ΔER+ΔEU+EUt+PE+UPE<bold>+PE*EUt+UPE*EUt</bold></p><p>6. repeat ∼ 1+ΔER+ΔEU+EUt+PE+UPE<bold>+PE*ΔEU+PE*EUt+UPE*ΔEU+UPE*EUt</bold></p><p>*p &lt; 0.05, **p &lt; 0.01, ***p &lt; 0.001.</p></caption>
<graphic xlink:href="587016v2_figs9.tif" mime-subtype="tiff" mimetype="image"/>
</fig>
<fig id="figs10" position="float" orientation="portrait" fig-type="figure">
<label>Figure S10.</label>
<caption><title>Decorrelation of reward and uncertainty by forced choices.</title>
<p>Anticorrelation (Pearson’s <italic>r</italic>) of ΔER and ΔEU on each free trial following a forced choice period. This analysis reveals the benefit of interleaving periods of free and forced choices: a negative correlation between ΔER and ΔEU builds up within periods of free trials, which forced choices (temporarily) abolish.</p></caption>
<graphic xlink:href="587016v2_figs10.tif" mime-subtype="tiff" mimetype="image"/>
</fig>
<fig id="figs11" position="float" orientation="portrait" fig-type="figure">
<label>Figure S11.</label>
<caption><title>Ablation tests with full vs partial cross-validation.</title>
<p>Decrements in model fit relative to model with all main effects (<italic>repeat ∼ 1+ΔER+ΔEU+EUt+PE+UPE</italic>), from removing each additional factor (UPE, PE, EUt, ΔEU, left to right), for “Main models”, reported in the text (excluding volatility) and with “Full crossvalidation” (including volatility parameter). This analysis shows that similar conclusions can be made when fitting all parameters (including volatility) compared to the simpler version adopted in the main text, where volatility is not fitted in each fold of the crossvalidation.</p></caption>
<graphic xlink:href="587016v2_figs11.tif" mime-subtype="tiff" mimetype="image"/>
</fig>
<fig id="figs12" position="float" orientation="portrait" fig-type="figure">
<label>Figure S12.</label>
<caption><title>Instructions for illustrated version of the task with back story (translated from French to English).</title></caption>
<graphic xlink:href="587016v2_figs12.tif" mime-subtype="tiff" mimetype="image"/>
<graphic xlink:href="587016v2_figs12a.tif" mime-subtype="tiff" mimetype="image"/>
</fig>
</sec>
<ref-list>
<title>References</title>
<ref id="c1"><label>1.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Kembro</surname>, <given-names>J. M.</given-names></string-name>, <string-name><surname>Lihoreau</surname>, <given-names>M.</given-names></string-name>, <string-name><surname>Garriga</surname>, <given-names>J.</given-names></string-name>, <string-name><surname>Raposo</surname>, <given-names>E. P.</given-names></string-name> &amp; <string-name><surname>Bartumeus</surname>, <given-names>F</given-names></string-name></person-group>. <article-title>Bumblebees learn foraging routes through exploitation–exploration cycles</article-title>. <source>J. R. Soc. Interface</source> <volume>16</volume>, <fpage>20190103</fpage> (<year>2019</year>).</mixed-citation></ref>
<ref id="c2"><label>2.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Denison</surname>, <given-names>S.</given-names></string-name>, <string-name><surname>Bonawitz</surname>, <given-names>E.</given-names></string-name>, <string-name><surname>Gopnik</surname>, <given-names>A.</given-names></string-name> &amp; <string-name><surname>Griffiths</surname>, <given-names>T. L</given-names></string-name></person-group>. <article-title>Rational variability in children’s causal inferences: The Sampling Hypothesis</article-title>. <source>Cognition</source> <volume>126</volume>, <fpage>285</fpage>–<lpage>300</lpage> (<year>2013</year>).</mixed-citation></ref>
<ref id="c3"><label>3.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Schulz</surname>, <given-names>E.</given-names></string-name> <etal>et al.</etal></person-group> <article-title>Structured, uncertainty-driven exploration in real-world consumer choice</article-title>. <source>Proc. Natl. Acad. Sci</source>. <volume>116</volume>, <fpage>13903</fpage>–<lpage>13908</lpage> (<year>2019</year>).</mixed-citation></ref>
<ref id="c4"><label>4.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Giron</surname>, <given-names>A. P.</given-names></string-name> <etal>et al.</etal></person-group> <article-title>Developmental changes in exploration resemble stochastic optimization</article-title>. <source>Nat. Hum. Behav</source>. (<year>2023</year>) doi:<pub-id pub-id-type="doi">10.1038/s41562-023-01662-1</pub-id>.</mixed-citation></ref>
<ref id="c5"><label>5.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Cohen</surname>, <given-names>J. D.</given-names></string-name>, <string-name><surname>McClure</surname>, <given-names>S. M.</given-names></string-name> &amp; <string-name><surname>Yu</surname>, <given-names>A. J</given-names></string-name></person-group>. <article-title>Should I stay or should I go? How the human brain manages the trade-off between exploitation and exploration</article-title>. <source>Philos. Trans. R. Soc. B Biol. Sci</source>. <volume>362</volume>, <fpage>933</fpage>–<lpage>942</lpage> (<year>2007</year>).</mixed-citation></ref>
<ref id="c6"><label>6.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Gittins</surname>, <given-names>J. C</given-names></string-name></person-group>. <article-title>Bandit processes and dynamic allocation indices</article-title>. <source>J. R. Stat. Soc. Ser. B Stat. Methodol</source>. <volume>41</volume>, <fpage>148</fpage>–<lpage>164</lpage> (<year>1979</year>).</mixed-citation></ref>
<ref id="c7"><label>7.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Cogliati Dezza</surname>, <given-names>I.</given-names></string-name>, <string-name><surname>Yu</surname>, <given-names>A. J.</given-names></string-name>, <string-name><surname>Cleeremans</surname>, <given-names>A.</given-names></string-name> &amp; <string-name><surname>Alexander</surname>, <given-names>W.</given-names></string-name></person-group> <article-title>Learning the value of information and reward over time when solving exploration-exploitation problems</article-title>. <source>Sci. Rep</source>. <volume>7</volume>, <fpage>16919</fpage> (<year>2017</year>).</mixed-citation></ref>
<ref id="c8"><label>8.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Gershman</surname>, <given-names>S. J</given-names></string-name></person-group>. <article-title>Deconstructing the human algorithms for exploration</article-title>. <source>Cognition</source> <volume>173</volume>, <fpage>34</fpage>–<lpage>42</lpage> (<year>2018</year>).</mixed-citation></ref>
<ref id="c9"><label>9.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Gershman</surname>, <given-names>S. J.</given-names></string-name></person-group> <article-title>Uncertainty and exploration</article-title>. <source>Decision</source> <volume>6</volume>, <fpage>277</fpage>–<lpage>286</lpage> (<year>2019</year>).</mixed-citation></ref>
<ref id="c10"><label>10.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Speekenbrink</surname>, <given-names>M.</given-names></string-name> &amp; <string-name><surname>Konstantinidis</surname>, <given-names>E</given-names></string-name></person-group>. <article-title>Uncertainty and Exploration in a Restless Bandit Problem</article-title>. <source>Top. Cogn. Sci</source>. <volume>7</volume>, <fpage>351</fpage>–<lpage>367</lpage> (<year>2015</year>).</mixed-citation></ref>
<ref id="c11"><label>11.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Wilson</surname>, <given-names>R. C.</given-names></string-name>, <string-name><surname>Geana</surname>, <given-names>A.</given-names></string-name>, <string-name><surname>White</surname>, <given-names>J. M.</given-names></string-name>, <string-name><surname>Ludvig</surname>, <given-names>E. A.</given-names></string-name> &amp; <string-name><surname>Cohen</surname>, <given-names>J. D</given-names></string-name></person-group>. <article-title>Humans use directed and random exploration to solve the explore–exploit dilemma</article-title>. <source>J. Exp. Psychol. Gen</source>. <volume>143</volume>, <fpage>2074</fpage>–<lpage>2081</lpage> (<year>2014</year>).</mixed-citation></ref>
<ref id="c12"><label>12.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Wu</surname>, <given-names>C. M.</given-names></string-name>, <string-name><surname>Schulz</surname>, <given-names>E.</given-names></string-name>, <string-name><surname>Speekenbrink</surname>, <given-names>M.</given-names></string-name>, <string-name><surname>Nelson</surname>, <given-names>J. D.</given-names></string-name> &amp; <string-name><surname>Meder</surname>, <given-names>B</given-names></string-name></person-group>. <article-title>Generalization guides human exploration in vast decision spaces</article-title>. <source>Nat. Hum. Behav</source>. <volume>2</volume>, <fpage>915</fpage>–<lpage>924</lpage> (<year>2018</year>).</mixed-citation></ref>
<ref id="c13"><label>13.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Cockburn</surname>, <given-names>J.</given-names></string-name>, <string-name><surname>Man</surname>, <given-names>V.</given-names></string-name>, <string-name><surname>Cunningham</surname>, <given-names>W. A.</given-names></string-name> &amp; <string-name><surname>O’Doherty</surname>, <given-names>J. P</given-names></string-name></person-group>. <article-title>Novelty and uncertainty regulate the balance between exploration and exploitation through distinct mechanisms in the human brain</article-title>. <source>Neuron</source> <volume>110</volume>, <fpage>2691</fpage>–<lpage>2702.e8</lpage> (<year>2022</year>).</mixed-citation></ref>
<ref id="c14"><label>14.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Payzan-LeNestour</surname>, <given-names>E.</given-names></string-name> &amp; <string-name><surname>Bossaerts</surname>, <given-names>P.</given-names></string-name></person-group> <article-title>Risk, Unexpected Uncertainty, and Estimation Uncertainty: Bayesian Learning in Unstable Settings</article-title>. <source>PLoS Comput. Biol</source>. <volume>7</volume>, <fpage>e1001048</fpage> (<year>2011</year>).</mixed-citation></ref>
<ref id="c15"><label>15.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Badre</surname>, <given-names>D.</given-names></string-name>, <string-name><surname>Doll</surname>, <given-names>B. B.</given-names></string-name>, <string-name><surname>Long</surname>, <given-names>N. M.</given-names></string-name> &amp; <string-name><surname>Frank</surname>, <given-names>M. J</given-names></string-name></person-group>. <article-title>Rostrolateral Prefrontal Cortex and Individual Differences in Uncertainty-Driven Exploration</article-title>. <source>Neuron</source> <volume>73</volume>, <fpage>595</fpage>–<lpage>607</lpage> (<year>2012</year>).</mixed-citation></ref>
<ref id="c16"><label>16.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Daw</surname>, <given-names>N. D.</given-names></string-name>, <string-name><surname>O’Doherty</surname>, <given-names>J. P.</given-names></string-name>, <string-name><surname>Dayan</surname>, <given-names>P.</given-names></string-name>, <string-name><surname>Seymour</surname>, <given-names>B.</given-names></string-name> &amp; <string-name><surname>Dolan</surname>, <given-names>R. J</given-names></string-name></person-group>. <article-title>Cortical substrates for exploratory decisions in humans</article-title>. <source>Nature</source> <volume>441</volume>, <fpage>876</fpage>–<lpage>879</lpage> (<year>2006</year>).</mixed-citation></ref>
<ref id="c17"><label>17.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Cogliati Dezza</surname>, <given-names>I.</given-names></string-name>, <string-name><surname>Noel</surname>, <given-names>X.</given-names></string-name>, <string-name><surname>Cleeremans</surname>, <given-names>A.</given-names></string-name> &amp; <string-name><surname>Yu</surname>, <given-names>A. J.</given-names></string-name></person-group> <article-title>Distinct motivations to seek out information in healthy individuals and problem gamblers</article-title>. <source>Transl. Psychiatry</source> <volume>11</volume>, <fpage>408</fpage> (<year>2021</year>).</mixed-citation></ref>
<ref id="c18"><label>18.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Dubois</surname>, <given-names>M.</given-names></string-name> &amp; <string-name><surname>Hauser</surname>, <given-names>T. U</given-names></string-name></person-group>. <article-title>Value-free random exploration is linked to impulsivity</article-title>. <source>Nat. Commun</source>. <volume>13</volume>, <fpage>4542</fpage> (<year>2022</year>).</mixed-citation></ref>
<ref id="c19"><label>19.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Fan</surname>, <given-names>H.</given-names></string-name>, <string-name><surname>Gershman</surname>, <given-names>S. J.</given-names></string-name> &amp; <string-name><surname>Phelps</surname>, <given-names>E. A</given-names></string-name></person-group>. <article-title>Trait somatic anxiety is associated with reduced directed exploration and underestimation of uncertainty</article-title>. <source>Nat. Hum. Behav</source>. <volume>7</volume>, <fpage>102</fpage>–<lpage>113</lpage> (<year>2022</year>).</mixed-citation></ref>
<ref id="c20"><label>20.</label><mixed-citation publication-type="confproc"><person-group person-group-type="author"><string-name><surname>Guo</surname>, <given-names>D.</given-names></string-name> &amp; <string-name><surname>Yu</surname>, <given-names>A. J.</given-names></string-name></person-group> <article-title>Revisiting the Role of Uncertainty-Driven Exploration in a (Perceived) Non-Stationary World</article-title>. in <conf-name>CogSci… Annual Conference of the Cognitive Science Society. Cognitive Science Society (US)</conf-name>. vol. <volume>43</volume> <fpage>2045</fpage> (<publisher-name>NIH Public Access</publisher-name>, <year>2021</year>).</mixed-citation></ref>
<ref id="c21"><label>21.</label><mixed-citation publication-type="book"><person-group person-group-type="author"><string-name><surname>Cogliati Dezza</surname>, <given-names>I.</given-names></string-name>, <string-name><surname>Schulz</surname>, <given-names>E.</given-names></string-name> &amp; <string-name><surname>Wu</surname>, <given-names>C. M.</given-names></string-name></person-group> <source>The Drive for Knowledge</source>. (<publisher-name>Cambridge University Press</publisher-name>, <year>2022</year>).</mixed-citation></ref>
<ref id="c22"><label>22.</label><mixed-citation publication-type="confproc"><person-group person-group-type="author"><string-name><surname>Machado</surname>, <given-names>M. C.</given-names></string-name>, <string-name><surname>Bellemare</surname>, <given-names>M. G.</given-names></string-name> &amp; <string-name><surname>Bowling</surname>, <given-names>M</given-names></string-name></person-group>. <article-title>Count-based exploration with the successor representation</article-title>. in <conf-name>Proceedings of the AAAI Conference on Artificial Intelligence</conf-name> vol. <volume>34</volume> <fpage>5125</fpage>–<lpage>5133</lpage> (<year>2020</year>).</mixed-citation></ref>
<ref id="c23"><label>23.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Palminteri</surname>, <given-names>S.</given-names></string-name>, <string-name><surname>Wyart</surname>, <given-names>V.</given-names></string-name> &amp; <string-name><surname>Koechlin</surname>, <given-names>E</given-names></string-name></person-group>. <article-title>The importance of falsification in computational cognitive modeling</article-title>. <source>Trends Cogn. Sci</source>. <volume>21</volume>, <fpage>425</fpage>–<lpage>433</lpage> (<year>2017</year>).</mixed-citation></ref>
<ref id="c24"><label>24.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Toyama</surname>, <given-names>A.</given-names></string-name>, <string-name><surname>Katahira</surname>, <given-names>K.</given-names></string-name> &amp; <string-name><surname>Kunisato</surname>, <given-names>Y</given-names></string-name></person-group>. <article-title>Examinations of Biases by Model Misspecification and Parameter Reliability of Reinforcement Learning Models</article-title>. <source>Comput. Brain Behav</source>. <volume>6</volume>, <fpage>651</fpage>– <lpage>670</lpage> (<year>2023</year>).</mixed-citation></ref>
<ref id="c25"><label>25.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Aarts</surname>, <given-names>H.</given-names></string-name>, <string-name><surname>Verplanken</surname>, <given-names>B.</given-names></string-name> &amp; <string-name><surname>Van Knippenberg</surname>, <given-names>A</given-names></string-name></person-group>. <article-title>Predicting Behavior From Actions in the Past: Repeated Decision Making or a Matter of Habit?</article-title> <source>J. Appl. Soc. Psychol</source>. <volume>28</volume>, <fpage>1355</fpage>– <lpage>1374</lpage> (<year>1998</year>).</mixed-citation></ref>
<ref id="c26"><label>26.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Akaishi</surname>, <given-names>R.</given-names></string-name>, <string-name><surname>Umeda</surname>, <given-names>K.</given-names></string-name>, <string-name><surname>Nagase</surname>, <given-names>A.</given-names></string-name> &amp; <string-name><surname>Sakai</surname>, <given-names>K</given-names></string-name></person-group>. <article-title>Autonomous mechanism of internal choice estimate underlies decision inertia</article-title>. <source>Neuron</source> <volume>81</volume>, <fpage>195</fpage>–<lpage>206</lpage> (<year>2014</year>).</mixed-citation></ref>
<ref id="c27"><label>27.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Balcarras</surname>, <given-names>M.</given-names></string-name>, <string-name><surname>Ardid</surname>, <given-names>S.</given-names></string-name>, <string-name><surname>Kaping</surname>, <given-names>D.</given-names></string-name>, <string-name><surname>Everling</surname>, <given-names>S.</given-names></string-name> &amp; <string-name><surname>Womelsdorf</surname>, <given-names>T</given-names></string-name></person-group>. <article-title>Attentional selection can be predicted by reinforcement learning of task-relevant stimulus features weighted by value-independent stickiness</article-title>. <source>J. Cogn. Neurosci</source>. <volume>28</volume>, <fpage>333</fpage>–<lpage>349</lpage> (<year>2016</year>).</mixed-citation></ref>
<ref id="c28"><label>28.</label><mixed-citation publication-type="book"><person-group person-group-type="author"><string-name><surname>Breland</surname>, <given-names>K.</given-names></string-name> &amp; <string-name><surname>Breland</surname>, <given-names>M</given-names></string-name></person-group>. <source>Animal behavior</source>. (<year>1966</year>). <publisher-name>Macmillan</publisher-name></mixed-citation></ref>
<ref id="c29"><label>29.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Dayan</surname>, <given-names>P.</given-names></string-name>, <string-name><surname>Niv</surname>, <given-names>Y.</given-names></string-name>, <string-name><surname>Seymour</surname>, <given-names>B.</given-names></string-name> &amp; <string-name><surname>Daw</surname>, <given-names>N. D</given-names></string-name></person-group>. <article-title>The misbehavior of value and the discipline of the will</article-title>. <source>Neural Netw</source>. <volume>19</volume>, <fpage>1153</fpage>–<lpage>1160</lpage> (<year>2006</year>).</mixed-citation></ref>
<ref id="c30"><label>30.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Dickinson</surname>, <given-names>A.</given-names></string-name> &amp; <string-name><surname>Balleine</surname>, <given-names>B</given-names></string-name></person-group>. <article-title>The role of learning in the operation of motivational systems</article-title>. <source>Stevens’ Handb. Exp. Psychol</source>. <volume>3</volume>, <fpage>497</fpage>–<lpage>533</lpage> (<year>2002</year>).</mixed-citation></ref>
<ref id="c31"><label>31.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Gershman</surname>, <given-names>S. J.</given-names></string-name>, <string-name><surname>Markman</surname>, <given-names>A. B.</given-names></string-name> &amp; <string-name><surname>Otto</surname>, <given-names>A. R</given-names></string-name></person-group>. <article-title>Retrospective revaluation in sequential decision making: a tale of two systems</article-title>. <source>J. Exp. Psychol. Gen</source>. <volume>143</volume>, <fpage>182</fpage> (<year>2014</year>).</mixed-citation></ref>
<ref id="c32"><label>32.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Guitart-Masip</surname>, <given-names>M.</given-names></string-name> <etal>et al.</etal></person-group> <article-title>Go and no-go learning in reward and punishment: interactions between affect and effect</article-title>. <source>Neuroimage</source> <volume>62</volume>, <fpage>154</fpage>–<lpage>166</lpage> (<year>2012</year>).</mixed-citation></ref>
<ref id="c33"><label>33.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Lee</surname>, <given-names>D.</given-names></string-name>, <string-name><surname>McGreevy</surname>, <given-names>B. P.</given-names></string-name> &amp; <string-name><surname>Barraclough</surname>, <given-names>D. J</given-names></string-name></person-group>. <article-title>Learning and decision making in monkeys during a rock–paper–scissors game</article-title>. <source>Cogn. Brain Res</source>. <volume>25</volume>, <fpage>416</fpage>–<lpage>430</lpage> (<year>2005</year>).</mixed-citation></ref>
<ref id="c34"><label>34.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Padoa-Schioppa</surname>, <given-names>C</given-names></string-name></person-group>. <article-title>Neuronal Origins of Choice Variability in Economic Decisions</article-title>. <source>Neuron</source> <volume>80</volume>, <fpage>1322</fpage>–<lpage>1336</lpage> (<year>2013</year>).</mixed-citation></ref>
<ref id="c35"><label>35.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Urai</surname>, <given-names>A. E.</given-names></string-name> &amp; <string-name><surname>Donner</surname>, <given-names>T. H</given-names></string-name></person-group>. <article-title>Persistent activity in human parietal cortex mediates perceptual choice repetition bias</article-title>. <source>Nat. Commun</source>. <volume>13</volume>, <fpage>6015</fpage> (<year>2022</year>).</mixed-citation></ref>
<ref id="c36"><label>36.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Gershman</surname>, <given-names>S. J</given-names></string-name></person-group>. <article-title>Origin of perseveration in the trade-off between reward and complexity</article-title>. <source>Cognition</source> <volume>204</volume>, <fpage>104394</fpage> (<year>2020</year>).</mixed-citation></ref>
<ref id="c37"><label>37.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Gershman</surname>, <given-names>S. J.</given-names></string-name>, <string-name><surname>Horvitz</surname>, <given-names>E. J.</given-names></string-name> &amp; <string-name><surname>Tenenbaum</surname>, <given-names>J. B</given-names></string-name></person-group>. <article-title>Computational rationality: A converging paradigm for intelligence in brains, minds, and machines</article-title>. <source>Science</source> <volume>349</volume>, <fpage>273</fpage>–<lpage>278</lpage> (<year>2015</year>).</mixed-citation></ref>
<ref id="c38"><label>38.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Kool</surname>, <given-names>W.</given-names></string-name> &amp; <string-name><surname>Botvinick</surname>, <given-names>M</given-names></string-name></person-group>. <article-title>The intrinsic cost of cognitive control</article-title>. <source>Behav. Brain Sci</source>. <volume>36</volume>, <fpage>697</fpage>– <lpage>698</lpage> (<year>2013</year>).</mixed-citation></ref>
<ref id="c39"><label>39.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Lieder</surname>, <given-names>F.</given-names></string-name> &amp; <string-name><surname>Griffiths</surname>, <given-names>T. L.</given-names></string-name></person-group> <article-title>When to use which heuristic: A rational solution to the strategy selection problem</article-title>. in <source>CogSci</source> (<year>2015</year>).</mixed-citation></ref>
<ref id="c40"><label>40.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Lieder</surname>, <given-names>F.</given-names></string-name> &amp; <string-name><surname>Griffiths</surname>, <given-names>T. L</given-names></string-name></person-group>. <article-title>Resource-rational analysis: Understanding human cognition as the optimal use of limited computational resources</article-title>. <source>Behav. Brain Sci</source>. <volume>43</volume>, <fpage>e1</fpage> (<year>2020</year>).</mixed-citation></ref>
<ref id="c41"><label>41.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Walker</surname>, <given-names>E. Y.</given-names></string-name> <etal>et al.</etal></person-group> <article-title>Studying the neural representations of uncertainty</article-title>. <source>Nat. Neurosci</source>. <volume>26</volume>, <fpage>1857</fpage>–<lpage>1867</lpage> (<year>2023</year>).</mixed-citation></ref>
<ref id="c42"><label>42.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Yu</surname>, <given-names>A. J.</given-names></string-name> &amp; <string-name><surname>Dayan</surname>, <given-names>P.</given-names></string-name></person-group> <article-title>Uncertainty, neuromodulation, and attention</article-title>. <source>Neuron</source> <volume>46</volume>, <fpage>681</fpage>–<lpage>692</lpage> (<year>2005</year>).</mixed-citation></ref>
<ref id="c43"><label>43.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Soltani</surname>, <given-names>A.</given-names></string-name> &amp; <string-name><surname>Izquierdo</surname>, <given-names>A</given-names></string-name></person-group>. <article-title>Adaptive learning under expected and unexpected uncertainty</article-title>. <source>Nat. Rev. Neurosci</source>. <volume>20</volume>, <fpage>635</fpage>–<lpage>644</lpage> (<year>2019</year>).</mixed-citation></ref>
<ref id="c44"><label>44.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Siegel</surname>, <given-names>S.</given-names></string-name> &amp; <string-name><surname>Allan</surname>, <given-names>L. G</given-names></string-name></person-group>. <article-title>The widespread influence of the Rescorla-Wagner model</article-title>. <source>Psychon. Bull. Rev</source>. <volume>3</volume>, <fpage>314</fpage>–<lpage>321</lpage> (<year>1996</year>).</mixed-citation></ref>
<ref id="c45"><label>45.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Miller</surname>, <given-names>R. R.</given-names></string-name>, <string-name><surname>Barnet</surname>, <given-names>R. C.</given-names></string-name> &amp; <string-name><surname>Grahame</surname>, <given-names>N. J</given-names></string-name></person-group>. <article-title>Assessment of the Rescorla-Wagner model</article-title>. <source>Psychol. Bull</source>. <volume>117</volume>, <fpage>363</fpage> (<year>1995</year>).</mixed-citation></ref>
<ref id="c46"><label>46.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Soto</surname>, <given-names>F. A.</given-names></string-name>, <string-name><surname>Vogel</surname>, <given-names>E. H.</given-names></string-name>, <string-name><surname>Uribe-Bahamonde</surname>, <given-names>Y. E.</given-names></string-name> &amp; <string-name><surname>Perez</surname>, <given-names>O. D</given-names></string-name></person-group>. <article-title>Why is the Rescorla-Wagner model so influential?</article-title> <source>Neurobiol. Learn. Mem</source>. <volume>204</volume>, <fpage>107794</fpage> (<year>2023</year>).</mixed-citation></ref>
<ref id="c47"><label>47.</label><mixed-citation publication-type="book"><person-group person-group-type="author"><string-name><surname>Sutton</surname>, <given-names>R. S.</given-names></string-name> &amp; <string-name><surname>Barto</surname>, <given-names>A. G</given-names></string-name></person-group>. <source>Reinforcement Learning: An Introduction</source>. (<publisher-name>MIT press</publisher-name>, <year>2018</year>).</mixed-citation></ref>
<ref id="c48"><label>48.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Filipowicz</surname>, <given-names>A. L.</given-names></string-name>, <string-name><surname>Glaze</surname>, <given-names>C. M.</given-names></string-name>, <string-name><surname>Kable</surname>, <given-names>J. W.</given-names></string-name> &amp; <string-name><surname>Gold</surname>, <given-names>J. I</given-names></string-name></person-group>. <article-title>Pupil diameter encodes the idiosyncratic, cognitive complexity of belief updating</article-title>. <source>Elife</source> <volume>9</volume>, <elocation-id>e57872</elocation-id> (<year>2020</year>).</mixed-citation></ref>
<ref id="c49"><label>49.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Nassar</surname>, <given-names>M. R.</given-names></string-name>, <string-name><surname>Wilson</surname>, <given-names>R. C.</given-names></string-name>, <string-name><surname>Heasly</surname>, <given-names>B.</given-names></string-name> &amp; <string-name><surname>Gold</surname>, <given-names>J. I</given-names></string-name></person-group>. <article-title>An approximately Bayesian delta-rule model explains the dynamics of belief updating in a changing environment</article-title>. <source>J. Neurosci</source>. <volume>30</volume>, <fpage>12366</fpage>–<lpage>12378</lpage> (<year>2010</year>).</mixed-citation></ref>
<ref id="c50"><label>50.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Zhang</surname>, <given-names>S.</given-names></string-name> &amp; <string-name><surname>Yu</surname>, <given-names>A. J</given-names></string-name></person-group>. <article-title>Forgetful Bayes and myopic planning: Human learning and decision-making in a bandit setting</article-title>. <source>Adv. Neural Inf. Process. Syst</source>. <volume>26</volume>, (<year>2013</year>).</mixed-citation></ref>
<ref id="c51"><label>51.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Auer</surname>, <given-names>P</given-names></string-name></person-group>. <article-title>Using Confidence Bounds for Exploitation-Exploration Trade-offs</article-title>. <source>J. Mach. Learn. Res</source>. (<year>2002</year>).</mixed-citation></ref>
<ref id="c52"><label>52.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Daw</surname>, <given-names>N. D.</given-names></string-name>, <string-name><surname>Niv</surname>, <given-names>Y.</given-names></string-name> &amp; <string-name><surname>Dayan</surname>, <given-names>P</given-names></string-name></person-group>. <article-title>Uncertainty-based competition between prefrontal and dorsolateral striatal systems for behavioral control</article-title>. <source>Nat. Neurosci</source>. <volume>8</volume>, <fpage>1704</fpage>–<lpage>1711</lpage> (<year>2005</year>).</mixed-citation></ref>
<ref id="c53"><label>53.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Lee</surname>, <given-names>S. W.</given-names></string-name>, <string-name><surname>Shimojo</surname>, <given-names>S.</given-names></string-name> &amp; <string-name><surname>O’Doherty</surname>, <given-names>J. P</given-names></string-name></person-group>. <article-title>Neural computations underlying arbitration between model-based and model-free learning</article-title>. <source>Neuron</source> <volume>81</volume>, <fpage>687</fpage>–<lpage>699</lpage> (<year>2014</year>).</mixed-citation></ref>
<ref id="c54"><label>54.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Addicott</surname>, <given-names>M. A.</given-names></string-name> <etal>et al.</etal></person-group> <article-title>Attention-deficit/hyperactivity disorder and the explore/exploit trade-off</article-title>. <source>Neuropsychopharmacology</source> <volume>46</volume>, <fpage>614</fpage>–<lpage>621</lpage> (<year>2021</year>).</mixed-citation></ref>
<ref id="c55"><label>55.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Wiehler</surname>, <given-names>A.</given-names></string-name>, <string-name><surname>Chakroun</surname>, <given-names>K.</given-names></string-name> &amp; <string-name><surname>Peters</surname>, <given-names>J</given-names></string-name></person-group>. <article-title>Attenuated directed exploration during reinforcement learning in gambling disorder</article-title>. <source>J. Neurosci</source>. <volume>41</volume>, <fpage>2512</fpage>–<lpage>2522</lpage> (<year>2021</year>).</mixed-citation></ref>
<ref id="c56"><label>56.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Averbeck</surname>, <given-names>B. B.</given-names></string-name> <etal>et al.</etal></person-group> <article-title>Uncertainty about mapping future actions into rewards may underlie performance on multiple measures of impulsivity in behavioral addiction: evidence from Parkinson’s disease</article-title>. <source>Behav. Neurosci</source>. <volume>127</volume>, <fpage>245</fpage> (<year>2013</year>).</mixed-citation></ref>
<ref id="c57"><label>57.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Djamshidian</surname>, <given-names>A.</given-names></string-name>, <string-name><surname>O’Sullivan</surname>, <given-names>S. S.</given-names></string-name>, <string-name><surname>Wittmann</surname>, <given-names>B. C.</given-names></string-name>, <string-name><surname>Lees</surname>, <given-names>A. J.</given-names></string-name> &amp; <string-name><surname>Averbeck</surname>, <given-names>B. B</given-names></string-name></person-group>. <article-title>Novelty seeking behaviour in Parkinson’s disease</article-title>. <source>Neuropsychologia</source> <volume>49</volume>, <fpage>2483</fpage>–<lpage>2488</lpage> (<year>2011</year>).</mixed-citation></ref>
<ref id="c58"><label>58.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Charpentier</surname>, <given-names>C. J.</given-names></string-name>, <string-name><surname>Aylward</surname>, <given-names>J.</given-names></string-name>, <string-name><surname>Roiser</surname>, <given-names>J. P.</given-names></string-name> &amp; <string-name><surname>Robinson</surname>, <given-names>O. J</given-names></string-name></person-group>. <article-title>Enhanced risk aversion, but not loss aversion, in unmedicated pathological anxiety</article-title>. <source>Biol. Psychiatry</source> <volume>81</volume>, <fpage>1014</fpage>–<lpage>1022</lpage> (<year>2017</year>).</mixed-citation></ref>
<ref id="c59"><label>59.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Bennett</surname>, <given-names>D.</given-names></string-name>, <string-name><surname>Sutcliffe</surname>, <given-names>K.</given-names></string-name>, <string-name><surname>Tan</surname>, <given-names>N. P.-J.</given-names></string-name>, <string-name><surname>Smillie</surname>, <given-names>L. D.</given-names></string-name> &amp; <string-name><surname>Bode</surname>, <given-names>S</given-names></string-name></person-group>. <article-title>Anxious and obsessive-compulsive traits are independently associated with valuation of noninstrumental information</article-title>. <source>J. Exp. Psychol. Gen</source>. <volume>150</volume>, <fpage>739</fpage> (<year>2021</year>).</mixed-citation></ref>
<ref id="c60"><label>60.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Grupe</surname>, <given-names>D. W.</given-names></string-name> &amp; <string-name><surname>Nitschke</surname>, <given-names>J. B</given-names></string-name></person-group>. <article-title>Uncertainty and anticipation in anxiety: an integrated neurobiological and psychological perspective</article-title>. <source>Nat. Rev. Neurosci</source>. <volume>14</volume>, <fpage>488</fpage>–<lpage>501</lpage> (<year>2013</year>).</mixed-citation></ref>
<ref id="c61"><label>61.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Aberg</surname>, <given-names>K. C.</given-names></string-name>, <string-name><surname>Toren</surname>, <given-names>I.</given-names></string-name> &amp; <string-name><surname>Paz</surname>, <given-names>R</given-names></string-name></person-group>. <article-title>A neural and behavioral trade-off between value and uncertainty underlies exploratory decisions in normative anxiety</article-title>. <source>Mol. Psychiatry</source> <volume>27</volume>, <fpage>1573</fpage>– <lpage>1587</lpage> (<year>2022</year>).</mixed-citation></ref>
<ref id="c62"><label>62.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Dayan</surname>, <given-names>P.</given-names></string-name> &amp; <string-name><surname>Balleine</surname>, <given-names>B. W.</given-names></string-name></person-group> <article-title>Reward, motivation, and reinforcement learning</article-title>. <source>Neuron</source> <volume>36</volume>, <fpage>285</fpage>–<lpage>298</lpage> (<year>2002</year>).</mixed-citation></ref>
<ref id="c63"><label>63.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Wood</surname>, <given-names>W.</given-names></string-name> &amp; <string-name><surname>Rünger</surname>, <given-names>D.</given-names></string-name></person-group> <article-title>Psychology of Habit</article-title>. <source>Annu. Rev. Psychol</source>. <volume>67</volume>, <fpage>289</fpage>–<lpage>314</lpage> (<year>2016</year>).</mixed-citation></ref>
<ref id="c64"><label>64.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Miller</surname>, <given-names>K. J.</given-names></string-name>, <string-name><surname>Shenhav</surname>, <given-names>A.</given-names></string-name> &amp; <string-name><surname>Ludvig</surname>, <given-names>E. A.</given-names></string-name></person-group> <article-title>Habits without values</article-title>. <source>Psychol. Rev</source>. <volume>126</volume>, <fpage>292</fpage> (<year>2019</year>).</mixed-citation></ref>
<ref id="c65"><label>65.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Adams</surname>, <given-names>C. D.</given-names></string-name> &amp; <string-name><surname>Dickinson</surname>, <given-names>A</given-names></string-name></person-group>. <article-title>Instrumental Responding following Reinforcer Devaluation</article-title>. <source>Q. J. Exp. Psychol. Sect. B</source> <volume>33</volume>, <fpage>109</fpage>–<lpage>121</lpage> (<year>1981</year>).</mixed-citation></ref>
<ref id="c66"><label>66.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Dorfman</surname>, <given-names>H. M.</given-names></string-name> &amp; <string-name><surname>Gershman</surname>, <given-names>S. J</given-names></string-name></person-group>. <article-title>Controllability governs the balance between Pavlovian and instrumental action selection</article-title>. <source>Nat. Commun</source>. <volume>10</volume>, <fpage>5826</fpage> (<year>2019</year>).</mixed-citation></ref>
<ref id="c67"><label>67.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Guitart-Masip</surname>, <given-names>M.</given-names></string-name>, <string-name><surname>Duzel</surname>, <given-names>E.</given-names></string-name>, <string-name><surname>Dolan</surname>, <given-names>R.</given-names></string-name> &amp; <string-name><surname>Dayan</surname>, <given-names>P</given-names></string-name></person-group>. <article-title>Action versus valence in decision making</article-title>. <source>Trends Cogn. Sci</source>. <volume>18</volume>, <fpage>194</fpage>–<lpage>202</lpage> (<year>2014</year>).</mixed-citation></ref>
<ref id="c68"><label>68.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Williams</surname>, <given-names>D. R.</given-names></string-name> &amp; <string-name><surname>Williams</surname>, <given-names>H</given-names></string-name></person-group>. <article-title>AUTO-MAINTENANCE IN THE PIGEON: SUSTAINED PECKING DESPITE CONTINGENT NON-REINFORCEMENT</article-title> <sup><xref ref-type="bibr" rid="c2">2</xref></sup>. <source>J. Exp. Anal. Behav</source>. <volume>12</volume>, <fpage>511</fpage>–<lpage>520</lpage> (<year>1969</year>).</mixed-citation></ref>
<ref id="c69"><label>69.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Brown</surname>, <given-names>P. L.</given-names></string-name> &amp; <string-name><surname>Jenkins</surname>, <given-names>H. M</given-names></string-name></person-group>. <article-title>AUTO-SHAPING OF THE PIGEON’S KEY-PECK 1</article-title>. <source>J. Exp. Anal. Behav</source>. <volume>11</volume>, <fpage>1</fpage>–<lpage>8</lpage> (<year>1968</year>).</mixed-citation></ref>
<ref id="c70"><label>70.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Hershberger</surname>, <given-names>W. A</given-names></string-name></person-group>. <article-title>An approach through the looking-glass</article-title>. <source>Anim. Learn. Behav</source>. <volume>14</volume>, <fpage>443</fpage>– <lpage>451</lpage> (<year>1986</year>).</mixed-citation></ref>
<ref id="c71"><label>71.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Findling</surname>, <given-names>C.</given-names></string-name>, <string-name><surname>Skvortsova</surname>, <given-names>V.</given-names></string-name>, <string-name><surname>Dromnelle</surname>, <given-names>R.</given-names></string-name>, <string-name><surname>Palminteri</surname>, <given-names>S.</given-names></string-name> &amp; <string-name><surname>Wyart</surname>, <given-names>V</given-names></string-name></person-group>. <article-title>Computational noise in reward-guided learning drives behavioral variability in volatile environments</article-title>. <source>Nat. Neurosci</source>. <volume>22</volume>, <fpage>2066</fpage>–<lpage>2077</lpage> (<year>2019</year>).</mixed-citation></ref>
<ref id="c72"><label>72.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Findling</surname>, <given-names>C.</given-names></string-name> &amp; <string-name><surname>Wyart</surname>, <given-names>V</given-names></string-name></person-group>. <article-title>Computation noise in human learning and decision-making: origin, impact, function</article-title>. <source>Curr. Opin. Behav. Sci</source>. <volume>38</volume>, <fpage>124</fpage>–<lpage>132</lpage> (<year>2021</year>).</mixed-citation></ref>
<ref id="c73"><label>73.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Wyart</surname>, <given-names>V.</given-names></string-name> &amp; <string-name><surname>Koechlin</surname>, <given-names>E</given-names></string-name></person-group>. <article-title>Choice variability and suboptimality in uncertain environments</article-title>. <source>Curr. Opin. Behav. Sci</source>. <volume>11</volume>, <fpage>109</fpage>–<lpage>115</lpage> (<year>2016</year>).</mixed-citation></ref>
<ref id="c74"><label>74.</label><mixed-citation publication-type="book"><person-group person-group-type="author"><string-name><surname>Sutton</surname>, <given-names>R. S.</given-names></string-name> &amp; <string-name><surname>Barto</surname>, <given-names>A. G.</given-names></string-name></person-group> <source>Introduction to Reinforcement Learning</source>. vol. <volume>135</volume> (<publisher-name>MIT press Cambridge</publisher-name>, <year>1998</year>).</mixed-citation></ref>
<ref id="c75"><label>75.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Ashwood</surname>, <given-names>Z. C.</given-names></string-name> <etal>et al.</etal></person-group> <article-title>Mice alternate between discrete strategies during perceptual decision-making</article-title>. <source>Nat. Neurosci</source>. <volume>25</volume>, <fpage>201</fpage>–<lpage>212</lpage> (<year>2022</year>).</mixed-citation></ref>
<ref id="c76"><label>76.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Weiss</surname>, <given-names>M. D.</given-names></string-name></person-group> <article-title>Weiss functional impairment rating scale (WFIRS) self-report</article-title>. <source>Vanc. Can. Univ. Br. Columbia Retrieved Naceonline ComAdultADHDtoolkitassessmenttoolswfirs Pdf</source> (<year>2000</year>).</mixed-citation></ref>
<ref id="c77"><label>77.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Micoulaud-Franchi</surname>, <given-names>J.-A.</given-names></string-name> <etal>et al.</etal></person-group> <article-title>Validation of the French Version of the Weiss Functional Impairment Rating Scale–Self-Report in a Large Cohort of Adult Patients With ADHD</article-title>. <source>J. Atten. Disord</source>. <volume>23</volume>, <fpage>1148</fpage>–<lpage>1159</lpage> (<year>2019</year>).</mixed-citation></ref>
<ref id="c78"><label>78.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Bruchon-Schweitzer</surname>, <given-names>M.</given-names></string-name> &amp; <string-name><surname>Paulhan</surname>, <given-names>I</given-names></string-name></person-group>. <article-title>Manuel de l’inventaire d’Anxiété trait-état (forme Y)</article-title>. <source>Lab. Ed Fr</source>. (<year>1990</year>).</mixed-citation></ref>
<ref id="c79"><label>79.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Scheier</surname>, <given-names>M. F.</given-names></string-name>, <string-name><surname>Carver</surname>, <given-names>C. S.</given-names></string-name> &amp; <string-name><surname>Bridges</surname>, <given-names>M. W</given-names></string-name></person-group>. <article-title>Distinguishing optimism from neuroticism (and trait anxiety, self-mastery, and self-esteem): a reevaluation of the Life Orientation Test</article-title>. <source>J. Pers. Soc. Psychol</source>. <volume>67</volume>, <fpage>1063</fpage> (<year>1994</year>).</mixed-citation></ref>
<ref id="c80"><label>80.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Trottier</surname>, <given-names>C.</given-names></string-name>, <string-name><surname>Mageau</surname>, <given-names>G.</given-names></string-name>, <string-name><surname>Trudel</surname>, <given-names>P.</given-names></string-name> &amp; <string-name><surname>Halliwell</surname>, <given-names>W. R</given-names></string-name></person-group>. <article-title>Validation de la version canadienne-française du Life Orientation Test-Revised</article-title>. <source>Can. J. Behav. Sci. Can. Sci. Comport</source>. <volume>40</volume>, <fpage>238</fpage> (<year>2008</year>).</mixed-citation></ref>
<ref id="c81"><label>81.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Patton</surname>, <given-names>J. H.</given-names></string-name>, <string-name><surname>Stanford</surname>, <given-names>M. S.</given-names></string-name> &amp; <string-name><surname>Barratt</surname>, <given-names>E. S</given-names></string-name></person-group>. <article-title>Factor structure of the barratt impulsiveness scale</article-title>. <source>J. Clin. Psychol</source>. <volume>51</volume>, <fpage>768</fpage>–<lpage>774</lpage> (<year>1995</year>).</mixed-citation></ref>
<ref id="c82"><label>82.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Bayle</surname>, <given-names>F. J.</given-names></string-name> <etal>et al.</etal></person-group> <article-title>Factor analysis of french translation of the Barratt impulsivity scale (BIS-10)</article-title>. <source>Can. J. Psychiatry Rev. Can. Psychiatr</source>. <volume>45</volume>, <fpage>156</fpage>–<lpage>165</lpage> (<year>2000</year>).</mixed-citation></ref>
<ref id="c83"><label>83.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Snaith</surname>, <given-names>R. P.</given-names></string-name> <etal>et al.</etal></person-group> <article-title>A scale for the assessment of hedonic tone the Snaith–Hamilton Pleasure Scale</article-title>. <source>Br. J. Psychiatry</source> <volume>167</volume>, <fpage>99</fpage>–<lpage>103</lpage> (<year>1995</year>).</mixed-citation></ref>
<ref id="c84"><label>84.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Gaillard</surname>, <given-names>R.</given-names></string-name>, <string-name><surname>Gourion</surname>, <given-names>D.</given-names></string-name> &amp; <string-name><surname>Llorca</surname>, <given-names>P. M.</given-names></string-name></person-group> <article-title>L’anhédonie dans la dépression</article-title>. <source>L’encéphale</source> <volume>39</volume>, <fpage>296</fpage>–<lpage>305</lpage> (<year>2013</year>).</mixed-citation></ref>
<ref id="c85"><label>85.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Baron-Cohen</surname>, <given-names>S.</given-names></string-name>, <string-name><surname>Wheelwright</surname>, <given-names>S.</given-names></string-name>, <string-name><surname>Skinner</surname>, <given-names>R.</given-names></string-name>, <string-name><surname>Martin</surname>, <given-names>J.</given-names></string-name> &amp; <string-name><surname>Clubley</surname>, <given-names>E.</given-names></string-name></person-group> <article-title>The autism-spectrum quotient (AQ): evidence from Asperger syndrome/high-functioning autism, males and females, scientists and mathematicians</article-title>. <source>J. Autism Dev. Disord</source>. <volume>31</volume>, <fpage>5</fpage>–<lpage>17</lpage> (<year>2001</year>).</mixed-citation></ref>
<ref id="c86"><label>86.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Kempenaers</surname>, <given-names>C.</given-names></string-name>, <string-name><surname>Braun</surname>, <given-names>S.</given-names></string-name>, <string-name><surname>Delvaux</surname>, <given-names>N.</given-names></string-name> &amp; <string-name><surname>Linkowski</surname>, <given-names>P</given-names></string-name></person-group>. <article-title>The assessment of autistic traits with the Autism Spectrum Quotient: Contribution of the French version to its construct validity</article-title>. <source>Eur. Rev. Appl. Psychol</source>. <volume>67</volume>, <fpage>299</fpage>–<lpage>306</lpage> (<year>2017</year>).</mixed-citation></ref>
<ref id="c87"><label>87.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Raine</surname>, <given-names>A</given-names></string-name></person-group>. <article-title>The SPQ: a scale for the assessment of schizotypal personality based on DSM-III-R criteria</article-title>. <source>Schizophr. Bull</source>. <volume>17</volume>, <fpage>555</fpage>–<lpage>564</lpage> (<year>1991</year>).</mixed-citation></ref>
<ref id="c88"><label>88.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Dumas</surname>, <given-names>P.</given-names></string-name>, <string-name><surname>Rosenfeld</surname>, <given-names>F.</given-names></string-name>, <string-name><surname>Saoud</surname>, <given-names>M.</given-names></string-name>, <string-name><surname>Dalery</surname>, <given-names>J.</given-names></string-name> &amp; <string-name><surname>d’Amato</surname>, <given-names>T.</given-names></string-name></person-group> <article-title>Translation and French adaptation of the Raine Schizotypal personality questionnaire</article-title>. <source>L’encephale</source> <volume>25</volume>, <fpage>315</fpage>–<lpage>322</lpage> (<year>1999</year>).</mixed-citation></ref>
<ref id="c89"><label>89.</label><mixed-citation publication-type="book"><person-group person-group-type="author"><string-name><surname>Goldberg</surname>, <given-names>L. R</given-names></string-name></person-group>. <chapter-title>An alternative “description of personality”: The Big-Five factor structure</chapter-title>. In <source>Personality and Personality Disorders</source> <fpage>34</fpage>–<lpage>47</lpage> (<publisher-name>Routledge</publisher-name>, <year>2013</year>).</mixed-citation></ref>
<ref id="c90"><label>90.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Plaisant</surname>, <given-names>O.</given-names></string-name>, <string-name><surname>Courtois</surname>, <given-names>R.</given-names></string-name>, <string-name><surname>Réveillère</surname>, <given-names>C.</given-names></string-name>, <string-name><surname>Mendelsohn</surname>, <given-names>G. A.</given-names></string-name> &amp; <string-name><surname>John</surname>, <given-names>O. P.</given-names></string-name></person-group> <article-title>Validation par analyse factorielle du Big Five Inventory français (BFI-Fr). Analyse convergente avec le NEO-PI-R</article-title>.  <source>Annales Médico-psychologiques, revue psychiatrique</source> vol. <volume>168</volume> <fpage>97</fpage>–<lpage>106</lpage> (<publisher-name>Elsevier</publisher-name>, <year>2010</year>).</mixed-citation></ref>
<ref id="c91"><label>91.</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><surname>Rescorla</surname>, <given-names>R. A</given-names></string-name></person-group>. <article-title>A theory of Pavlovian conditioning: Variations in the effectiveness of reinforcement and non-reinforcement</article-title>. <source>Class. Cond. Curr. Res. Theory</source> <volume>2</volume>, <fpage>64</fpage>–<lpage>69</lpage> (<year>1972</year>).</mixed-citation></ref>
</ref-list>
</back>
<sub-article id="sa0" article-type="editor-report">
<front-stub>
<article-id pub-id-type="doi">10.7554/eLife.103363.1.sa3</article-id>
<title-group>
<article-title>eLife Assessment</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Diaconescu</surname>
<given-names>Andreea Oliviana</given-names>
</name>
<role specific-use="editor">Reviewing Editor</role>
<aff>
<institution-wrap>
<institution>University of Toronto</institution>
</institution-wrap>
<city>Toronto</city>
<country>Canada</country>
</aff>
</contrib>
</contrib-group>
<kwd-group kwd-group-type="evidence-strength">
<kwd>Solid</kwd>
</kwd-group>
<kwd-group kwd-group-type="claim-importance">
<kwd>Valuable</kwd>
</kwd-group>
</front-stub>
<body>
<p>The findings of this study are <bold>valuable</bold>, as they address a critical methodological gap in decision-making research by demonstrating how heuristic strategies can confound interpretations of uncertainty-driven behaviour and provide a clearer framework for distinguishing between uncertainty-seeking and heuristic-driven exploration. While the evidence is <bold>solid</bold>, with strong methodological rigour in task design and computational modelling, some claims, such as the stability of uncertainty parameters and correlations with psychopathology measures, require refinement. Overall, the data broadly support the study's claims, but interpretational ambiguities limit the impact of certain findings.</p>
</body>
</sub-article>
<sub-article id="sa1" article-type="referee-report">
<front-stub>
<article-id pub-id-type="doi">10.7554/eLife.103363.1.sa2</article-id>
<title-group>
<article-title>Reviewer #1 (Public review):</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<anonymous/>
<role specific-use="referee">Reviewer</role>
</contrib>
</contrib-group>
</front-stub>
<body>
<p>Summary:</p>
<p>The study investigates how uncertainty and heuristic strategies influence reward-based decision-making, using a novel two-armed bandit task combined with computational modeling. It aims to disentangle uncertainty-driven behavior from heuristic strategies such as repetition bias and win-stay-lose-shift tendencies, while also exploring individual differences in these processes.</p>
<p>Strengths:</p>
<p>The paper is methodologically sound, and the inclusion of subjective reports enhances the validity of the model testing. The findings on the use of heuristics under specific uncertainty conditions are particularly intriguing.</p>
<p>Weaknesses:</p>
<p>(1) Unclear how the findings significantly diverge from previous work:</p>
<p>At the start of the introduction, the authors propose a working hypothesis of &quot;heterogeneity in the uncertainty effects.&quot; However, this concept is already well-established in the field. Foundational work by Yu and Dayan (2005) and more recent studies by Gershman and colleagues on total and relative uncertainty have provided substantial evidence supporting this idea. Additionally, the notion that such heterogeneity could explain mixed findings has been discussed in studies like Wilson (2014). What specific problem are the authors addressing here, and how does their work significantly differ from previous research?</p>
<p>Later on, however, it seems that the authors' hypothesis is to test the role of multiple factors in driving participants' decisions in the context considered by the authors. First, why is it important to solve such a puzzle? Second, this too has been investigated previously, see for example Dubois (2022), eLife. Therefore, what novel things is this paper bringing to the table? I do see that the task is novel - mostly combining different experimental strategies previously adopted - and that the model includes both heuristics and uncertainty-based strategies, which can account for their shared variance ... but are the authors really answering a novel question? Also, it is not very clear which question the authors are answering see point C below.</p>
<p>(2) The sample size appears to be quite small, and the results would be more convincing if supported by a replication study.</p>
<p>(3) The results section can be somewhat unclear at times, as it introduces novel aspects (e.g., the fMRI session) or questions that were not previously explained within the framework outlined in the introduction. While the findings related to psychopathology are interesting, their relevance to the research question posed in the introduction is not immediately clear. If these findings have significant added value, it would be helpful for the authors to highlight this earlier in the manuscript. Similarly, the results on individual differences in uncertainty (Section 3.6), though intriguing, appear tangential to the primary research question regarding the role of multiple factors in driving participants' decisions. Overall, it would strengthen the manuscript to clarify the main research question and ensure the results are more directly aligned with it.</p>
</body>
</sub-article>
<sub-article id="sa2" article-type="referee-report">
<front-stub>
<article-id pub-id-type="doi">10.7554/eLife.103363.1.sa1</article-id>
<title-group>
<article-title>Reviewer #2 (Public review):</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<anonymous/>
<role specific-use="referee">Reviewer</role>
</contrib>
</contrib-group>
</front-stub>
<body>
<p>Summary:</p>
<p>This paper addresses mixed findings regarding levels of uncertainty-seeking/avoidance in past reinforcement learning studies. Using computational modelling and a novel variant of a bandit task performed across two sessions, the authors investigate the extent to which uncertainty-driven behaviour can be distinguished from heuristic-like behaviours (e.g., repetition, win-stay/lose-switch). They demonstrate that heuristics account for a significant and stable portion of the variance in choice behaviour, which might otherwise be misattributed to uncertainty-driven parameters. Additionally, they find that relative uncertainty explains additional variance and provides some evidence of stability across sessions.</p>
<p>Strengths:</p>
<p>The task is well-designed to tease apart multiple different factors contributing to choice during a bandit task, including separating those tied to uncertainty per se versus other policies. They validate a Bayesian model to account for learning and choice behaviour, as well as subjective estimates of learned value and confidence in these values. The work employs comprehensive model comparison to characterise behaviour in this task, and points to important risks within research on uncertainty preferences using bandit-like tasks when failing to fully account for heuristic-like drivers of such behaviour.</p>
<p>Weaknesses:</p>
<p>Part of this work seeks to relate individual differences in various choice parameters across sessions and to relate those to self-report scales. The estimates of cross-session reliability are valuable, particularly when comparing across the different parameters (e.g., heuristic ones being most robust), but the uncertainty-related parameters are interpreted too liberally (i.e., as being stable across sessions when both were weak and one was not significant). Moreover, the correlations with external scales are very hard to interpret given the number of comparisons that were run without correction. The findings overall will have value to people interested in modelling uncertainty preferences in learning tasks -- some of whom have considered heuristic factors less than others -- but perhaps be of more moderate impact beyond this group.</p>
</body>
</sub-article>
<sub-article id="sa3" article-type="referee-report">
<front-stub>
<article-id pub-id-type="doi">10.7554/eLife.103363.1.sa0</article-id>
<title-group>
<article-title>Reviewer #3 (Public review):</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<anonymous/>
<role specific-use="referee">Reviewer</role>
</contrib>
</contrib-group>
</front-stub>
<body>
<p>Summary:</p>
<p>This work investigated how uncertainty, repetition bias, and win-stay-lose-shift processes influence reward-based decision-making. Using a modified two-armed bandit task and computational models, the authors provide evidence for individual variation in the integration of uncertainty on choice behaviour that remains somewhat stable across two experiment sessions. The authors also find a number of interesting results due to their ability to disentangle components of this decision-making process using their novel task and models. Specifically, they find that higher total uncertainty leads people to use more heuristic-based strategies like making repetitive choices or engaging in win-stay-lose-shift behaviour. However, they also find that there are individual differences in how people use uncertainty to guide their choices, and that these differences are consistent within individuals across multiple experiment sessions. This finding can help explain prior inconsistencies in the literature, where researchers have found evidence for both uncertainty-seeking and uncertainty-avoidance tendencies. Overall, this research adds to our understanding of the mechanisms of uncertainty-modulated learning and decision-making.</p>
<p>Strengths:</p>
<p>One of the primary strengths of this research is that it helps provide support for the idea that mixed and null results in the prior literature could be due to individual differences in uncertainty preferences and that this individual variation is somewhat stable within subjects across multiple experiment sessions. The authors cleverly disentangle expected reward and uncertainty by interleaving free and forced choice trials in their behavioural task, illuminating the novel impact of reward and uncertainty on this particular decision process. However, it should be noted that this behavioural decorrelation does not persist beyond the first few trials after a forced choice period, so whether or not the decorrelation is truly robust remains unclear.</p>
<p>The authors also use computational modelling to further probe the influence of uncertainty on reward-based choices. Specifically, they compare a Bayesian ideal observer learning model and a variation on a standard Rescorla-Wagner model, finding that a version of the Bayesian model fits the participants' behaviour best. The model descriptions and analyses are clearly explained and methodologically rigorous.</p>
<p>Interestingly, the authors find that both repetition bias and model parameters that capture a win-stay-lose-shift strategy (signed and unsigned previous prediction error) significantly improve their model fits. They also make an important point that if win-stay-lose-shift behaviour is not controlled for, then switch behaviour (for example, switching to a lower expected reward option after receiving a large loss) may appear to be uncertainty-seeking when it is not. This idea speaks to a larger point that future research should be careful to not conflate &quot;exploration&quot; with &quot;uncertainty-seeking.&quot;</p>
<p>Weaknesses:</p>
<p>This research has some weaknesses regarding the correlations between the psychopathology measures and the computational model parameters. First, the choice of self-report measures is not well supported by any specific hypotheses. Relatedly, the authors do not include sufficient rationale for their choice to include only results from the anxiety and impulsivity measures in the main text while leaving out significant findings for a number of correlations between other measures and parameter coefficients. It is also not clear how the model parameters are being derived for use in each of these correlational analyses. In sum, the manuscript as-is contains inconsistent and/or confusing reporting of correlation results that require further clarification.</p>
</body>
</sub-article>
</article>