<?xml version="1.0" ?><!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Archiving and Interchange DTD v1.3 20210610//EN"  "JATS-archivearticle1-mathml3.dtd"><article xmlns:ali="http://www.niso.org/schemas/ali/1.0/" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article" dtd-version="1.3" xml:lang="en">
<front>
<journal-meta>
<journal-id journal-id-type="nlm-ta">elife</journal-id>
<journal-id journal-id-type="publisher-id">eLife</journal-id>
<journal-title-group>
<journal-title>eLife</journal-title>
</journal-title-group>
<issn publication-format="electronic" pub-type="epub">2050-084X</issn>
<publisher>
<publisher-name>eLife Sciences Publications, Ltd</publisher-name>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">88514</article-id>
<article-id pub-id-type="doi">10.7554/eLife.88514</article-id>
<article-id pub-id-type="doi" specific-use="version">10.7554/eLife.88514.1</article-id>
<article-version-alternatives>
<article-version article-version-type="publication-state">reviewed preprint</article-version>
<article-version article-version-type="preprint-version">1.1</article-version>
</article-version-alternatives>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Neuroscience</subject>
</subj-group>
</article-categories>
<title-group>
<article-title>Inferring control objectives in a virtual balancing task in humans and monkeys</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author" corresp="yes">
<contrib-id contrib-id-type="orcid">http://orcid.org/0000-0003-2573-146X</contrib-id>
<name>
<surname>Sadeghi</surname>
<given-names>Mohsen</given-names>
</name>
<xref ref-type="aff" rid="a1">1</xref>
<xref ref-type="aff" rid="a2">2</xref>
<xref ref-type="author-notes" rid="n1">*</xref>
<xref ref-type="corresp" rid="cor1">%</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Razavian</surname>
<given-names>Reza Sharif</given-names>
</name>
<xref ref-type="aff" rid="a3">3</xref>
<xref ref-type="author-notes" rid="n1">*</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Bazzi</surname>
<given-names>Salah</given-names>
</name>
<xref ref-type="aff" rid="a1">1</xref>
<xref ref-type="aff" rid="a2">2</xref>
<xref ref-type="aff" rid="a4">4</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Chowdhury</surname>
<given-names>Raeed</given-names>
</name>
<xref ref-type="aff" rid="a5">5</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Batista</surname>
<given-names>Aaron</given-names>
</name>
<xref ref-type="aff" rid="a5">5</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Loughlin</surname>
<given-names>Patrick</given-names>
</name>
<xref ref-type="aff" rid="a5">5</xref>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Sternad</surname>
<given-names>Dagmar</given-names>
</name>
<xref ref-type="aff" rid="a1">1</xref>
<xref ref-type="aff" rid="a2">2</xref>
<xref ref-type="aff" rid="a4">4</xref>
<xref ref-type="aff" rid="a6">6</xref>
</contrib>
<aff id="a1"><label>1</label><institution>Departments of Biology, Northeastern University</institution>, Boston, MA, <country>USA</country></aff>
<aff id="a2"><label>2</label><institution>Departments of Electrical and Computer Engineering, Northeastern University</institution>, Boston, MA, <country>USA</country></aff>
<aff id="a3"><label>3</label>Northern Arizona University, AZ, USA</aff>
<aff id="a4"><label>4</label><institution>Institute for Experiential Robotics, Northeastern University</institution>, Boston, MA, <country>USA</country></aff>
<aff id="a5"><label>5</label><institution>Department of Bioengineering, and Center for the Neural Basis of Cognition, University of Pittsburgh</institution>, PA, <country>USA</country></aff>
<aff id="a6"><label>6</label><institution>Departments of Physics, Northeastern University</institution>, Boston, MA, <country>USA</country></aff>
</contrib-group>
<contrib-group content-type="section">
<contrib contrib-type="editor">
<name>
<surname>Cowan</surname>
<given-names>Noah J</given-names>
</name>
<role>Reviewing Editor</role>
<aff>
<institution-wrap>
<institution>Johns Hopkins University</institution>
</institution-wrap>
<city>Baltimore</city>
<country>United States of America</country>
</aff>
</contrib>
<contrib contrib-type="senior_editor">
<name>
<surname>Makin</surname>
<given-names>Tamar R</given-names>
</name>
<role>Senior Editor</role>
<aff>
<institution-wrap>
<institution>University of Cambridge</institution>
</institution-wrap>
<city>Cambridge</city>
<country>United Kingdom</country>
</aff>
</contrib>
</contrib-group>
<author-notes>
<corresp id="cor1"><label>%</label>Corresponding author; email: <email>m.sadeghi@northeastern.edu</email></corresp>
<fn fn-type="equal" id="n1"><label>*</label><p>Equal contribution</p></fn>
</author-notes>
<pub-date date-type="original-publication" iso-8601-date="2023-08-22">
<day>22</day>
<month>08</month>
<year>2023</year>
</pub-date>
<volume>12</volume>
<elocation-id>RP88514</elocation-id>
<history>
<date date-type="sent-for-review" iso-8601-date="2023-05-02">
<day>02</day>
<month>05</month>
<year>2023</year>
</date>
</history>
<pub-history>
<event>
<event-desc>Preprint posted</event-desc>
<date date-type="preprint" iso-8601-date="2023-05-02">
<day>02</day>
<month>05</month>
<year>2023</year>
</date>
<self-uri content-type="preprint" xlink:href="https://doi.org/10.1101/2023.05.02.539055"/>
</event>
</pub-history>
<permissions>
<copyright-statement>© 2023, Sadeghi et al</copyright-statement>
<copyright-year>2023</copyright-year>
<copyright-holder>Sadeghi et al</copyright-holder>
<ali:free_to_read/>
<license xlink:href="https://creativecommons.org/licenses/by/4.0/">
<ali:license_ref>https://creativecommons.org/licenses/by/4.0/</ali:license_ref>
<license-p>This article is distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution License</ext-link>, which permits unrestricted use and redistribution provided that the original author and source are credited.</license-p>
</license>
</permissions>
<self-uri content-type="pdf" xlink:href="elife-preprint-88514-v1.pdf"/>
<abstract>
<title>Abstract</title>
<p>Natural behaviors have redundancy, which implies that humans and animals can achieve their goals with different control strategies. Given only observations of behavior, is it possible to infer the control strategy that the subject is employing? This challenge is particularly acute in animal behavior because we cannot ask or instruct the subject to use a particular control strategy. This study presents a three-pronged approach to infer an animal’s control strategy from behavior. First, both humans and monkeys performed a virtual balancing task for which different control strategies could be utilized. Under matched experimental conditions, corresponding behaviors were observed in humans and monkeys. Second, a generative model was developed that identified two main control strategies to achieve the task goal. Model simulations were used to identify aspects of behavior that could distinguish which control strategy was being used. Third, these behavioral signatures allowed us to infer the control strategy used by human subjects who had been instructed to use one control strategy or the other. Based on this validation, we could then infer strategies from animal subjects. Being able to positively identify a subject’s control strategy from behavior can provide a powerful tool to neurophysiologists as they seek the neural mechanisms of sensorimotor coordination.</p>
</abstract>
<abstract abstract-type="teaser">
<title>Impact statement</title>
<p>A computational approach identifies control strategies in humans and monkeys to serve as basis for analysis of neural correlates of skillful manipulation.</p>
</abstract>

</article-meta>
<notes>
<notes notes-type="competing-interest-statement">
<title>Competing Interest Statement</title><p>The authors have declared no competing interest.</p></notes>
</notes>
</front>
<body>
<sec id="s1">
<title>Introduction</title>
<p>Almost all actions in daily life can be achieved in multiple ways that all can lead to the desired task goals. As an example, consider a driver steering a car on a curvy road. She may choose different paths depending on whether she wants to maintain a consistent distance from the median strip or whether she aims to minimize changes in velocity. Both strategies can take the driver to her destination, maybe even arriving at the same time, although the precise path taken by the car in both situations will differ. How could one identify the underlying control objectives from differences in observed behavior? A considerable number of studies in human movement neuroscience have aimed to identify the control strategies in a given task based on their kinematic manifestations (<xref ref-type="bibr" rid="c4">Braun et al., 2009</xref>; <xref ref-type="bibr" rid="c14">Izawa et al., 2008</xref>; <xref ref-type="bibr" rid="c22">Nagengast et al., 2009</xref>; <xref ref-type="bibr" rid="c32">Razavian et al., 2023</xref>; <xref ref-type="bibr" rid="c41">Uno et al., 1989</xref>; <xref ref-type="bibr" rid="c42">Wong et al., 2021</xref>). However, experimental tasks are often chosen to elicit consistent behavioral features across repetitions and individuals, not only to facilitate analysis, but also to constrain control to a single objective. Behavior in natural settings, however, is often complex and highly variable across repetitions, and individuals can employ a multitude of strategies to accomplish a task. To date, understanding of such variable behavior - let alone its neural bases - has posed formidable challenges (<xref ref-type="bibr" rid="c8">Croxson et al., 2009</xref>; <xref ref-type="bibr" rid="c11">Diedrichsen et al., 2010</xref>; <xref ref-type="bibr" rid="c18">Kawato, 1999</xref>; <xref ref-type="bibr" rid="c34">Scott, 2004</xref>).</p>
<p>Attempts to understand the neural underpinnings of control objectives have been pursued in research on both humans and non-human primates (<xref ref-type="bibr" rid="c3">Benyamini &amp; Zacksenhouse, 2015</xref>; <xref ref-type="bibr" rid="c7">Cross et al., 2023</xref>; <xref ref-type="bibr" rid="c8">Croxson et al., 2009</xref>; <xref ref-type="bibr" rid="c9">Desrochers et al., 2016</xref>; <xref ref-type="bibr" rid="c17">Kao et al., 2021</xref>; <xref ref-type="bibr" rid="c21">Miall et al., 2007</xref>; <xref ref-type="bibr" rid="c24">Nashed et al., 2014</xref>; <xref ref-type="bibr" rid="c26">Omrani et al., 2016</xref>). Yet, these two lines of inquiry have remained largely parallel with few direct bridges: human behavioral and computational research has mainly focused on the analysis of behavior, while animal research has used invasive methods such as intracortical recordings to gain direct insights into the neural mechanisms of movement control. Experiments with humans tend to use detailed experimental manipulations to elicit features of motor behavior that afford insights into its governing principles. Using a wide range of tasks, from simple reaching to interacting with complex objects, mathematical models with specific control algorithms have been used to reproduce the salient features of behavior (<xref ref-type="bibr" rid="c6">Crevecoeur et al., 2019</xref>; <xref ref-type="bibr" rid="c10">Diedrichsen, 2007</xref>; <xref ref-type="bibr" rid="c22">Nagengast et al., 2009</xref>; <xref ref-type="bibr" rid="c25">Nayeem et al., 2021</xref>; <xref ref-type="bibr" rid="c32">Razavian et al., 2023</xref>; <xref ref-type="bibr" rid="c43">Yeo et al., 2016</xref>). However, understanding the neural underpinnings of movement control at the intracortical level in healthy humans has remained a challenge. On the other hand, animal research, in particular with non-human primates, allows sophisticated methods to directly record neural activity to afford insights into neural correlates of motor behavior. Ultimately, this knowledge should transfer to how the human brain functions (<xref ref-type="bibr" rid="c1">Badre et al., 2015</xref>), but those links must be built.</p>
<p>To achieve this objective, cooperative study designs between human and animal motor research are needed to understand the neural basis of human motor skill (<xref ref-type="bibr" rid="c1">Badre et al., 2015</xref>; <xref ref-type="bibr" rid="c31">Rajalingham et al., 2022</xref>). However, there are difficult challenges to overcome: First, cooperative design requires matching behavioral tasks that can be performed similarly and with the same conditions by both humans and animals. The most appropriate animal model for many human behaviors are monkeys. Second, the goals and constraints of behavioral studies with monkeys and humans are somewhat different, which can preclude a direct comparison. Behavioral tasks used with monkeys are typically simpler than those used with humans, due to the animals’ more limited cognitive capacities. Also, studies with monkeys aim for highly repeatable behaviors to facilitate the examination of neural activity by aggregating it across trials or days. In contrast, studies of human behavior can push toward tasks that are more cognitively sophisticated and that capture the complexity that abounds in natural activities. This study bridges the gap between human and monkey behavioral studies to build toward an understanding of the neural principles of human motor control.</p>
<p>We used an experimental paradigm, the Critical Stability Task (CST), that can be performed by both humans and monkeys (<xref ref-type="bibr" rid="c30">Quick et al., 2018</xref>). The CST requires the subject to balance an unstable virtual system governed by a very simple dynamical equation (see Methods). Performing the task is akin to balancing a virtual pole. The CST has features that make it suitable for the study of more complex motor behaviors. First, while the goal remains the same, the difficulty of the task can be titrated. Second, it involves interactions with an object (albeit virtual in our case) so that continuous adjustments are required to succeed. Each trial evokes unique behavior that may reflect different control strategies to accomplish the task. In addition, even if the same control strategy is employed, each trial generates different behavior due to sensorimotor noise and the task’s instability. As in the car driving analogy, the subjects might seek to optimize position, or they might seek to optimize velocity, and different behavioral strategies may lead to equal success.</p>
<p>Because of its complexity and redundancy, each trial of the CST is unique. The goal of the study is to infer the subject’s control policy (i.e., optimize position or optimize velocity) from observations of their behavior. When the subjects are humans, it is possible to instruct them to employ a particular strategy or to ask them posthoc what strategy they adopted to succeed at the task. This explicit route is definitely not available with monkeys. As we are still quite far from ‘reading out’ strategies from neural activity, we need to start with behavior to infer the control strategies. Hence, this study adopted a computational approach based on optimal control theory to simulate behavior during the CST in various conditions. This approach allowed us to make predictions about the behavioral signatures associated with different control policies, which we then used to analyze the experimental data from both humans and monkeys.</p>
<p>In overview, this study investigated, through experimental data and model-based simulations, the sensorimotor origins of behavioral strategies in humans and non-human primates performing the CST. We developed the experimental paradigm such that humans and monkeys executed the task under matching conditions while recording movement kinematics in exactly the same way. An optimal control model was used to simulate different control objectives, through which we identified two different control strategies in the experimental data of humans and monkeys. We discuss how in the future these results could guide the analysis of neural data collected from monkeys to understand the neural underpinnings of different control policies in an interactive feedback-driven task with redundancy.</p>
</sec>
<sec id="s2">
<title>Results</title>
<p>The Critical Stability Task (CST) involved balancing an unstable system using horizontal movements of the hand to keep a cursor from moving off the screen (<bold><xref rid="fig1" ref-type="fig">Figure 1A, C</xref></bold>). This study collected data from human subjects performing the CST and compared it to previously collected data from monkeys performing the same task. The hand’s displacements were recorded by 3D motion capture (Qualisys, Gothenburg), with a reflective marker attached to the hand. The cursor dynamics were generated by a linear first-order dynamical system, relating hand and cursor kinematics as described in <xref ref-type="bibr" rid="c30">Quick et al., 2018</xref>:
<disp-formula id="eqn1">
<alternatives><graphic xlink:href="539055v1_eqn1.gif" mime-subtype="gif" mimetype="image"/></alternatives>
</disp-formula>
where <italic>x</italic> and <inline-formula><alternatives><inline-graphic xlink:href="539055v1_inline1.gif" mime-subtype="gif" mimetype="image"/></alternatives></inline-formula> are the horizontal cursor position and cursor velocity on the screen, <italic>p</italic> is the horizontal hand position, and <italic>λ</italic> is a positive constant fixed at the beginning of each trial. The parameter <italic>λ</italic> sets the gain of the system. When <italic>λ</italic> is larger, the cursor would tend to move faster, making the task more difficult as faster and more precise hand movements were required to maintain balance.</p>
<fig id="fig1" position="float" orientation="portrait" fig-type="figure">
<label>Figure 1:</label>
<caption><title>Experimental setup for monkeys and humans performing the CST.</title>
<p>Monkeys (<bold>A</bold>) and humans (<bold>C</bold>) controlled an unstable cursor displayed on a screen using lateral movements of their right hand. The hand movements were recorded using motion capture; the data were used in real-time to solve for the cursor position and velocity through the CST dynamics equation. Timeseries of the hand (red) and cursor (blue) movements shown for four example trials from monkeys (<bold>B</bold>) and humans (<bold>D</bold>).</p></caption>
<graphic xlink:href="539055v1_fig1.tif" mime-subtype="tiff" mimetype="image"/>
</fig>
<p>Correspondingly, success rates at the task decreased with increasing <italic>λ</italic>. To summarize the skill of human and monkey participants, we identified the value at which subjects succeeded at only 50% of the trials and defined that value as the “critical” <italic>λ</italic>.</p>
<p>The task goal was to keep the cursor within a range of space shown on the screen, i.e., −<italic>c</italic> ≤ <italic>x</italic>(<italic>t</italic>) ≤ <italic>c</italic>, where <italic>c</italic> was a positive constant. This created a redundancy in achieving the task goal as there were infinitely many ways in which one could balance the cursor inside the specified region. We examined movement kinematics to identify control strategies employed by different subjects, or across different trials.</p>
<p>In a previous study, two Rhesus monkeys were trained to perform the CST under increasing difficulty levels (<xref ref-type="bibr" rid="c30">Quick et al., 2018</xref>). Similarly, here 18 human subjects were recruited to perform the same task under comparable experimental conditions as the monkeys (see Methods). <bold><xref rid="fig1" ref-type="fig">Figure 1</xref></bold> illustrates the experimental setup for both monkeys and humans (<bold><xref rid="fig1" ref-type="fig">Figure 1A</xref> and <xref rid="fig1" ref-type="fig">1C</xref></bold>) and shows examples of their behavior (<bold><xref rid="fig1" ref-type="fig">Figure 1B</xref></bold> and <xref rid="fig1" ref-type="fig">1D</xref>). Overall, there were similarities in performance between humans and monkeys. To further quantify and compare this performance across humans and monkeys, we defined a set of control metrics to assess different aspects of control as detailed in the following.</p>
<sec id="s2a">
<title>Experiment 1: CST performance without instructed strategy</title>
<p>In the first experiment, six human subjects performed the CST with the only instruction to “perform the task without failing to the best of your ability”. Failure occurred if the cursor escaped the boundaries of the screen (±10cm from the center) within the trial duration of 6s. Subjects received categorical feedback about the outcome at the end of each trial in a text appearing on the screen reading “Well done!” for success, and “Failed!” for failure. The degree of difficulty, set by <italic>λ</italic>, was increased stepwise across trials until the subject could no longer perform the task (see Methods for the specifics about the setting of <italic>λ</italic> values).</p>
<p>We first sought to examine the main characteristics of behavior in CST performance and how it compared between humans and monkeys. To quantify the overall behavior, four main metrics were employed as described and motivated below. To begin, we considered the overall success rate in the task among different individuals, before focusing on the kinematics of task performance. <bold><xref rid="fig2" ref-type="fig">Figure 2A</xref></bold> illustrates the success rates and how they dropped as the task difficulty increased. Both humans and monkeys showed a similar pattern of decrease in success rate which was well-captured with a sigmoidal function. Expectedly, individuals varied in their ability to achieve high difficulty levels as a measure of skillful performance, indicated by their “critical <italic>λ</italic> value”, that is, the value of <italic>λ</italic> when the success rate drops below 50%. To investigate the performance in more detail, the kinematics of movement were examined, specifically the hand and cursor position during each trial. As indicated in <xref ref-type="disp-formula" rid="eqn1">equation (1)</xref>, the hand position <italic>p</italic> was the control input to the system which aimed to control the cursor position <italic>x</italic> as the variable of interest. Due to the unstable nature of the task, drifting of the cursor towards the edge of the screen demanded a response by a hand movement to avoid failure. As such, two simple metrics characterized control, one quantifying how the movement of hand and cursor correlated, and a second one to what degree the hand response lagged cursor displacements. <bold><xref rid="fig2" ref-type="fig">Figure 2B</xref></bold> shows the correlation between the cursor movement and the hand movement as a function of task difficulty. The strength of the correlation increased as trials became more challenging in both monkeys and humans, asymptoting towards –1. According to <xref ref-type="disp-formula" rid="eqn1">equation (1)</xref>, this behavior was equivalent to reducing the sum (<italic>p</italic>+ <italic>x</italic>) when <italic>λ</italic> increased, so as to prevent rapid changes in cursor velocity <inline-formula><alternatives><inline-graphic xlink:href="539055v1_inline2.gif" mime-subtype="gif" mimetype="image"/></alternatives></inline-formula>, and, hence, reduce the chance of failure.</p>
<fig id="fig2" position="float" orientation="portrait" fig-type="figure">
<label>Figure 2:</label>
<caption><title>Overall behavioral characteristics of CST performance as a function of task difficulty (λ).</title>
<p>Data is shown for two individual monkeys (first two columns from left) from a previous study (<xref ref-type="bibr" rid="c30">Quick et al., 2018</xref>), as well as an example human individual (third column from left) and the average across human subjects (right-most column). For the individual subjects, each data point and its corresponding error bars represent the mean±SD across trials for any given difficulty level, respectively. For the human average plot, the data points and their corresponding error bars represent the mean±SE across individuals for each difficulty level. <bold>A</bold>. Psychometric curves for success rate (%) as a function of task difficulty (λ) the difficulty level at which the success rate crossed 50% was considered as the critical stability point (λ<sub>c</sub>), indicating the individual’s skill level in task. <bold>B</bold>. Correlation between the hand and cursor movement during CST. <bold>C</bold>. Sensorimotor lag between the cursor and the hand movements. <bold>D</bold>. Ratio of hand RMS over the cursor RMS calculated for each trial, representing the gain of the response.</p></caption>
<graphic xlink:href="539055v1_fig2.tif" mime-subtype="tiff" mimetype="image"/>
</fig>
<p>The response lag from the cursor movement (observed feedback) to the hand movement (control response) is an important characteristic of a control system. As shown in <bold><xref rid="fig2" ref-type="fig">Figure 2C</xref></bold>, by increasing the task difficulty <italic>λ</italic>, the lag decreased for all subjects, meaning subjects generated faster corrective responses to cursor displacements in more difficult trials. A possible reason for such behavior is that higher <italic>λ</italic> values meant increased instability of the system, which required faster responses to avoid failure. Whereas in easy trials, due to slower dynamics of the system, subjects could afford delayed responses to cursor displacements (and hence, larger lags) and still manage to succeed.</p>
<p>As the fourth metric, we also calculated the control gain by measuring the ratio of root mean squared (RMS) of hand position to the RMS of cursor position for each trial. This measure determined to what extent the control signal (hand movement) compared in magnitude to the cursor movement. A large gain meant that on average across a trial, the hand exhibited larger movements than necessary to correct for cursor deviations. <bold><xref rid="fig2" ref-type="fig">Figure 2D</xref></bold> illustrates the calculated gain as a function of task difficulty for humans and monkeys. As shown, except for <italic>Monkey J</italic>, the gain showed a gradual decrease as the task difficulty increased for most individuals. Such decrease could be due to larger cursor movements at higher difficulty levels, and perhaps more efficient corrective hand responses to cursor displacements. To the latter, it is worth noting that for high λ values, small hand movements could cause large cursor displacements, which was detrimental to the task success. Therefore, pruning any task-irrelevant hand movements, consistent with promoting efficiency, seemed essential to succeed in more difficult trials. Overall, the control metrics presented in <bold><xref rid="fig2" ref-type="fig">Figure 2</xref></bold> give insight into how the CST was performed: as the task difficulty increased, subjects tended to respond to cursor displacements faster (that is, with lower lag), more precisely (seen in the stronger hand-cursor correlation), and more efficiently (with lower gain). Behavior was comparable between humans and monkeys, which suggests that there were underlying control strategies used in common by both species. Next, we sought to detect those control strategies.</p>
<sec id="s2a1">
<title>Redundancy of control strategies in CST performance</title>
<p>The CST, as described earlier, affords redundancy in the control strategies that could result in task success. Although covert in aggregate level of performance (i.e., <bold><xref rid="fig2" ref-type="fig">Figure 2</xref></bold>), single trial observations of hand and cursor movements suggested that different underlying control objectives might be at play. Two types of behavioral patterns appeared recognizable in the data. In one case, the cursor seemed to be always balanced around the center of the screen, and any deviations from the center induced a response to bring the cursor back to the center. This was reflected in the oscillatory movements of the cursor around the center, shown in example trials in <xref rid="fig1" ref-type="fig">Figure 1B</xref> and <xref rid="fig1" ref-type="fig">D</xref> (first row). In other trials, the cursor either exhibited a slow drift from the center or remained relatively still anywhere within the boundaries of the screen, with only limited attempts to bring the cursor back to the center (for example, <bold><xref rid="fig1" ref-type="fig">Figure 1B</xref></bold> and <xref rid="fig1" ref-type="fig">D</xref>, second row). We hypothesized that these patterns of behavior arise from different control objectives, each focused on a different state variable in the state-space of the cursor movement.</p>
<p>In the former case, the position of the cursor appeared to be the primary control variable. Under this strategy, subjects might pursue the objective of keeping the cursor near the center of the screen. We refer to this strategy as the Position Control strategy. In the latter case, the cursor velocity seemed to be of primary importance for control, with the objective to slow down cursor velocity regardless of its position in the workspace. We refer to this strategy as the Velocity Control strategy.</p>
<p>Can we distinguish between different control strategies by examining behavior? To test this idea, we took a computational approach by developing a generative model based on optimal feedback control (<xref ref-type="bibr" rid="c40">Todorov &amp; Jordan, 2002</xref>) that could simulate the task under different conditions and with different objectives (<xref ref-type="bibr" rid="c40">Todorov &amp; Jordan, 2002</xref>). The model involved a controller that generated optimal motor commands based on a given strategy to perform the CST via a simple effector model. The model also contained a state estimation block that estimated the states of the system based on the given feedback (<xref ref-type="bibr" rid="c39">Todorov, 2005</xref>). In this case, cursor position and cursor velocity were used as feedback to the controller at each time step. <bold><xref rid="fig3" ref-type="fig">Figure 3A</xref></bold> illustrates a block diagram of this model.</p>
<fig id="fig3" position="float" orientation="portrait" fig-type="figure">
<label>Figure 3:</label>
<caption><title>A generative model to performs the CST.</title>
<p><bold>A</bold>. An optimal feedback controller generates motor commands based on two control objectives, position and velocity control. The motor command leads the movement of the effector (hand), which performs the CST. The cursor position and velocity are provided as feedback from which all the states are estimated and fed back to the controller. <bold>B</bold> and <bold>C</bold>. Example trials simulated under the two control objectives for different difficulty levels: keeping the cursor at the center (<bold>B</bold>; position control) and keeping the cursor still (<bold>C</bold>; velocity control).</p></caption>
<graphic xlink:href="539055v1_fig3.tif" mime-subtype="tiff" mimetype="image"/>
</fig>
<p>The control gains used in the controller to generate the motor commands were optimally found by minimizing the sum of two cost functions: the cost of effort to reduce energy, as well as the cost of accuracy that prevented the states of the system from making large deviations <xref ref-type="disp-formula" rid="eqn2">(2)</xref>:
<disp-formula id="eqn2">
<alternatives><graphic xlink:href="539055v1_eqn2.gif" mime-subtype="gif" mimetype="image"/></alternatives>
</disp-formula>
where <italic>u</italic> and <bold>x</bold> represented the motor command and the state vector of the system, respectively. In this model, the state vector consisted of six states: the position, velocity and acceleration of the hand, as well as the position, velocity and acceleration of the cursor (see Methods). Variables <italic>t</italic> and <italic>n</italic> represent the time, and the total number of time steps, respectively, in a trial. The matrix <italic>Q</italic> and the scalar <italic>R</italic> determined the weight of accuracy and effort in the cost function, respectively. Importantly, the matrix <italic>Q</italic> allowed for determining which states of the system were of primary importance in the control process. Therefore, the implementation of different control objectives in the controller was done through setting the <italic>Q</italic> matrix appropriately. As such, a Position Control strategy was implemented by setting the weight of cursor position in the <italic>Q</italic> matrix to a large value, emphasizing the primacy of cursor position as a control variable. Similarly, to implement the Velocity Control strategy, the weight of the cursor velocity in the <italic>Q</italic> matrix was set to a large value (see Methods). By simulating the task for each control strategy, we could generate synthetic behavior similar to that of humans and monkeys. <bold><xref rid="fig3" ref-type="fig">Figure 3B</xref></bold> and <xref rid="fig3" ref-type="fig">C</xref> illustrate a few example simulations of the task under different difficulty levels for the Position Control and Velocity Control, respectively. As exemplified, the simulated trials for Position Control show oscillatory movements of the cursor around the center, whereas the trials generated based on Velocity Control, exhibited slow drift of the cursor from the center with minimal attempt to correct for such drift. These characteristics were similar to the observed patterns of behavior in human and monkey data (<xref rid="fig1" ref-type="fig">Figure 1B</xref> and <xref rid="fig1" ref-type="fig">D</xref>).</p>
<p>To further identify the behavioral signatures associated with each control objective, beyond the apparent differences between single trials, we conducted a series of simulations in which the model performance was examined for a range of task difficulties, and novel predictions of the model for each control objective were assessed. For each control objective, the task was simulated for different difficulty levels, ranging from <italic>λ</italic> = 1.5 to <italic>λ</italic> = 7, with increments of Δ<italic>λ</italic> = 0.2. For each difficulty level, 500 trials were simulated (see Methods for details). In the first step, we performed the same set of analyses as reported in <bold><xref rid="fig2" ref-type="fig">Figure 2</xref></bold> to evaluate how the model compared to human and monkey behavior at an aggregate level of CST performance. <bold><xref rid="fig4" ref-type="fig">Figure 4A</xref></bold> illustrates the overall performance of the model for both Position Control and Velocity Control strategies. As shown, for each metric, the model exhibited comparable behavior to experimental data with regard to the task difficulty: the success rate dropped in a sigmoidal fashion, the correlation between hand and cursor movements increased, and the response lag between hand and cursor as well as the hand/cursor gain decreased. These results showed that, overall, both simulated control strategies were capable of producing similar behavioral characteristics as humans and monkeys. But more interestingly, despite no apparent advantage of one strategy over the other in the task success (<bold><xref rid="fig4" ref-type="fig">Figure 4A</xref></bold>, top panel), they showed differences in the magnitude of hand-cursor correlation, lag and gain. Namely, Position Control consistently showed larger magnitudes for correlation, lag, and gain for any given task difficulty.</p>
<fig id="fig4" position="float" orientation="portrait" fig-type="figure">
<label>Figure 4:</label>
<caption><title>Different control objectives result in measurably different behavior.</title>
<p>Overall performance of the model (<bold>A</bold>) and human subjects (<bold>B</bold>) for two control objectives, Position Control and Velocity Control. The four rows show success rate (first row), correlation between hand and cursor movement (second row), sensorimotor lag between cursor and hand movements (third row), and the hand/cursor gain, defined as the RMS of hand movement over the RMS of cursor movement during each trial (last row). The error bars on the human average data indicate the standard error of the mean across subjects for each group. <bold>C</bold>. The average performance across difficulty levels and subjects within each group. The Critical λ (first row) indicates the difficulty level at which the success rate crosses 50%.</p></caption>
<graphic xlink:href="539055v1_fig4.tif" mime-subtype="tiff" mimetype="image"/>
</fig>
</sec>
</sec>
<sec id="s2b">
<title>Experiment 2: CST performance under explicit instructions</title>
<p>The model indicated that differences in behavioral metrics exist for Position vs Velocity Control. This led to a new experiment for which we recruited two new groups of human subjects (<italic>n</italic>=6 per group). Each group performed the CST under the same procedure as described in Experiment 1, except that this time each group was explicitly instructed to use a specific control strategy. One group was asked to perform the task with the objective of “keeping the cursor at the center of the screen at all times”. This instruction was to induce a Position Control strategy. The second group was asked to “keep the cursor still anywhere within the boundaries of the screen”. This instruction aimed to induce a Velocity Control strategy (see Methods for details). In each group, the kinematic behavior of hand and cursor was collected, and the control metrics were calculated. The goal was to elicit differences in performance between the two groups and, if such differences were found, determine whether they matched the behavior of the corresponding model.</p>
<p>The summary of performance for both human subject groups is shown in <bold><xref rid="fig4" ref-type="fig">Figure 4B</xref></bold>. The general trends of all four measures with respect to the task difficulty were consistent with the data generated by the model, as well as the human data from Experiment 1 (<bold><xref rid="fig2" ref-type="fig">Figure 2</xref></bold>). Importantly, the behavioral differences between the two control strategies in human data matched the predictions of the model relatively well (<bold><xref rid="fig4" ref-type="fig">Figure 4A, B</xref></bold>): the rate of success was similar, and with the exception of hand-cursor correlation, the group with Position Control instruction showed significantly larger hand-cursor lag (unpaired t-test: <italic>t</italic><sub>10</sub> = 3.79, <italic>p</italic> = 0.004) and hand-cursor gain (unpaired t-test: <italic>t</italic><sub>10</sub> = 5.27, <italic>p</italic> &lt; 10<sup>-3</sup>) compared to the group with Velocity Control instructions (<bold><xref rid="fig4" ref-type="fig">Figure 4C</xref></bold>).</p>
<p>These results showed that the model not only captured the overall performance features observed in the data, it also successfully demonstrated the redundancy of control strategies in CST performance, and qualitatively distinguished between such strategies at an aggregate level of performance. To ask further, can we identify, in a quantitative way, the control strategy employed by an individual, or even in a given trial, when no explicit information about their preferred strategy is available? To this end, we examined performance at single-trial level and introduced quantitative measures that evaluated the degree to which a particular control strategy was used in that trial, as described in the next section.</p>
<sec id="s2b1">
<title>Behavioral traces of control strategy in an individual’s overall performance</title>
<p>To further investigate what control strategy was preferred by an individual or in a given trial, we examined the predictions of the model about the cursor behavior in state space, and then tested these predictions using experimental data from Experiment 2. Two metrics were defined that captured the state-space behavior of the cursor in each trial. First, we examined the average cursor position and cursor velocity in each trial, represented in the state space of cursor movement. This provided a single data point for each trial in state space, indicating whether on average there was a drift in cursor position and its velocity away from zero (<inline-formula><alternatives><inline-graphic xlink:href="539055v1_inline3.gif" mime-subtype="gif" mimetype="image"/></alternatives></inline-formula>). It was expected that for Position Control, all trials scattered around the origin of the state space, whereas for Velocity Control, they could deviate from the origin. We also examined whether the states of the cursor correlated. <bold><xref rid="fig5" ref-type="fig">Figure 5A</xref></bold> illustrates the state-space representation of cursor movement based on model simulations for both Position Control (top) and Velocity Control (bottom), where each data point represents one simulated trial. As shown, the distribution of trials in this space differed markedly between the two control objectives. The Position Control strategy resulted in a distribution with little correlation between cursor position and its velocity, and closely scattered around the center. In contrast, the Velocity Control strategy revealed an elongated distribution with a relatively strong correlation between the cursor position and its velocity. This allowed us to distinguish between different individuals’ preferred control strategy.</p>
<fig id="fig5" position="float" orientation="portrait" fig-type="figure">
<label>Figure 5:</label>
<caption><title>State-space distribution of trials reveals different control strategies.</title>
<p><bold>A</bold>. Mean cursor velocity plotted against mean position for each trial, shown for the position control objective (top) and velocity control objective (bottom). Each data point represents one successful trial and was simulated for a range of difficulty levels up to the critical λ value (corresponding to 50% success rate). <bold>B</bold>. Three example human subjects from the position control group (top row) and velocity control group (bottom row). Each data point represents one successful trial. The data represents an ensemble of trials ranging in difficulty levels up to the critical λ value for each subject. R indicates the correlation between the trial position and velocities. <bold>C</bold>. Pearson correlation coefficient between cursor mean position and velocity for each control objective in the model (left) and human data (right). The human data shows the mean (±SE) across subjects for each control objective group.</p></caption>
<graphic xlink:href="539055v1_fig5.tif" mime-subtype="tiff" mimetype="image"/>
</fig>
<p>To validate the model predictions, the same analysis was performed on the empirical data from Experiment 2. <bold><xref rid="fig5" ref-type="fig">Figure 5B</xref></bold> illustrates three example subjects from Position Control and Velocity Control groups, and <bold><xref rid="fig5" ref-type="fig">Figure 5C</xref></bold> shows a summary of how the correlation values differed across control strategies for the model and the empirical data. As shown, overall, subjects in the Velocity Control group showed significantly larger correlations than individuals in the Position Control group (unpaired t-test on the Pearson correlation coefficient: <italic>t</italic><sub>10</sub> =4.06, <italic>p</italic>=0.002). Based on the within-group variability, this allowed us to determine how pronounced a subject executed their respective strategy compared to other subjects in the same group. This metric, therefore, provided a quantitative way of estimating where on the spectrum of control strategy an individual’s performance lies with respect to other performers.</p>
</sec>
<sec id="s2b2">
<title>The effects of control strategy at a single-trial level of behavior</title>
<p>Due to the task’s redundancy the choice of control strategy may not be fixed for an individual throughout their performance and might vary from one trial to the next. It is therefore of great interest to determine, in a given trial, to what extent the behavior is the outcome of Position versus Velocity Control strategies. To this end, we examined the magnitude of cursor movement calculated as the root mean squared (RMS) of its position and velocity in each trial. This was directly related to the objective functions used in the model (<xref ref-type="disp-formula" rid="eqn2">equation(2)</xref>, which provided a more direct comparison regarding the primacy of position versus velocity in the control of the cursor: a Position Control strategy aimed to minimize the RMS of cursor position, while Velocity Control aimed to minimize the RMS of cursor velocity. This distinction could be well represented in the state-space of the cursor movement.</p>
<p><bold><xref rid="fig6" ref-type="fig">Figure 6A</xref></bold> illustrates the model prediction for the RMS of cursor position and cursor velocity plotted against each other for the Position Control (top) and Velocity Control (bottom). For Position Control, the distribution of trials leans towards the vertical axis (restricting cursor position but allowing large cursor velocities), whereas for Velocity Control, it leans mainly towards the horizontal axis (a larger range of cursor positions but restricted velocities). This distinction could be quantified by the slope of a fitted regression line to the data, with relatively larger slopes indicating Position Control and smaller slopes signaling Velocity Control. Similar patterns of behavior could be observed in the human data from Experiment 2 as illustrated in <bold><xref rid="fig6" ref-type="fig">Figure 6B</xref></bold> and <xref rid="fig6" ref-type="fig">C</xref>, with the Position Control group showing significantly larger regression slope than the Velocity Control group (unpaired t-test, <italic>t</italic><sub>10</sub> = 6.33, <italic>p</italic>&lt;0.001). The regression slope could more clearly distinguish between individual trials than could the correlation coefficient metric shown in <bold><xref rid="fig5" ref-type="fig">Figure 5</xref></bold>, regarding their corresponding control strategy: if a given trial in the RMS space of the cursor movement lay below/above a certain slope threshold, its performance could be considered the result of a Velocity/Position Control strategy. We could therefore use this behavioral feature to develop a classifier that inferred, with a certain level of confidence, the underlying control strategy in the performance of an individual in any given trial.</p>
<fig id="fig6" position="float" orientation="portrait" fig-type="figure">
<label>Figure 6:</label>
<caption><title>Identifying control strategy based on magnitude of cursor movement in the state space.</title>
<p><bold>A</bold>. Magnitude of cursor movements quantified by the RMS of position and cursor velocity for each trial, plotted against each other; position control objective (top) and velocity control objective (bottom). Each data point represents one successful trial and was generated based on the model simulations for a range of difficulty levels up to the critical λ value (corresponding to 50% success rate). <bold>B</bold>. Performance of three example subjects from the position control group (top row) and velocity control group (bottom row). Each data point represents one successful trial. The data represents an ensemble of trials ranging in difficulty level up to the critical λ value for each subject. The values of the regression slopes are also shown. <bold>C</bold>. Summary of the regression slopes for the RMS plots, shown for each control objective in the model (left) and human data (right). The human data shows the mean (±SE) across subjects for each control objective group.</p></caption>
<graphic xlink:href="539055v1_fig6.tif" mime-subtype="tiff" mimetype="image"/>
</fig>
</sec>
<sec id="s2b3">
<title>Inferring control strategies from behavior during CST performance</title>
<p>When monkeys performed the CST, we lacked explicit knowledge about which strategy they might have employed. This is similar to Experiment 1; when humans performed the CST with no specific instructions, their control objective was not explicitly available. To achieve the goal of inferring an individual’s control objective based on their performance, we used the control characteristics that our computational approach introduced to distinguish between different control strategies. To this end, the simulation results based on the cursor movement in its RMS space (<bold><xref rid="fig6" ref-type="fig">Figure 6A</xref></bold>) were used to train a simple classifier, a support vector machine (see Methods). This classifier then determined, based on the learned regression slopes from the RMS distributions (<bold><xref rid="fig7" ref-type="fig">Figure 7A</xref></bold>), whether a given trial was likely performed under the Position Control, or Velocity Control strategy. We first tested the performance of the classifier on the empirical data from Experiment 2, where the intended control strategy used by each subject was known.</p>
<fig id="fig7" position="float" orientation="portrait" fig-type="figure">
<label>Figure 7:</label>
<caption><title>Classifying control strategies in humans who received explicit instructions.</title>
<p><bold>A</bold>. Simulated data in the RMS space of cursor movement used as training set for a classifier to determine the control objective of each trial. <bold>B</bold>. Data from three example subjects in each group, where each trial was classified as position control (brown), velocity control (cyan), or uncertain as to the control objective (grey). To obtain the control objective of each trial, the classifier (a support vector machine; see Methods) obtained the probability of that trial performed with position control objective, where P(pos)&gt;70% was classified as position control, P(pos)&lt;30% was classified as velocity control, and everything else was classified as uncertain. The average of P(pos) across all trials for each individual is shown inside the respective plot. <bold>C</bold>. Overall probability of Position Control summarized for all subjects instructed in the position and velocity control groups of Experiment 2.</p></caption>
<graphic xlink:href="539055v1_fig7.tif" mime-subtype="tiff" mimetype="image"/>
</fig>
<p><bold><xref rid="fig7" ref-type="fig">Figure 7B</xref></bold> shows the cursor RMS data from three example subjects in each instructed group (similar to <bold><xref rid="fig6" ref-type="fig">Figure 6</xref></bold>); for each trial (data point) a probability was obtained from the classifier indicating to what extent a given trial was performed with the Position Control strategy (see Methods). A trial with the estimated probability of &gt;70% was considered Position Control, while a probability of &lt;30% for a trial signified Velocity Control. All other probabilities were considered as ‘Uncertain’ as to which of the two control objectives were used. As shown in <bold><xref rid="fig7" ref-type="fig">Figure 7B</xref></bold>, for the Position Control group, most of the trials were rightfully classified as Position Control trials, and similarly for the Velocity Control group, the majority of trials were classified under Velocity Control strategy. The average probability across all trials for each individual was also obtained as an overall measure of the control objective for that subject. This average measure is shown in <bold><xref rid="fig7" ref-type="fig">Figure 7B</xref></bold> for the example subjects and summarized in <bold><xref rid="fig7" ref-type="fig">Figure 7C</xref></bold> for all subjects in each group. This showed that the classifier correctly determined the control strategy used by each individual without being trained on any experimental data.</p>
<p>The ultimate test of our approach would be to infer the control strategy used by individuals whose strategy was unknown, that is the monkeys, and humans who received no instructions about control strategy in Experiment 1. After representing the performance of each subject in the RMS space, the classifier was used to determine what control strategy was used in each trial. <bold><xref rid="fig8" ref-type="fig">Figure 8</xref></bold> illustrates the classification results for human subjects who received no instructions (Experiment 1) as well as two monkeys (<italic>Monkey I</italic> and <italic>J</italic> from Quick et. al. 2018). The model simulations are also provided as reference in <bold><xref rid="fig8" ref-type="fig">Figure 8A</xref></bold>. <bold><xref rid="fig8" ref-type="fig">Figure 8B</xref></bold> and C show the data from three example human subjects, as well as two monkeys, in which each trial is either labelled as Position Control (brown), Velocity Control (cyan), or Uncertain (grey). Two example trials, one from each inferred control strategy are also singled out from each subject’s performance in <bold><xref rid="fig8" ref-type="fig">Figure 8B</xref></bold> and C (bottom row) to show how the hand and cursor movement generally behaved under each control strategy. Calculating the average probability of control strategy for each individual, similar to <bold><xref rid="fig7" ref-type="fig">Figure 7</xref></bold>, we could infer which control strategy was of primary importance for each subject (<bold><xref rid="fig8" ref-type="fig">Figure 8D</xref></bold>). For example, human subject NI-S2 more likely adopted a Velocity control strategy, while human subject NI-S4 mainly performed the task with Position Control strategy (<bold><xref rid="fig8" ref-type="fig">Figure 8B</xref></bold>). Similarly, <italic>Monkey I</italic> seemed to prefer the Velocity Control strategy, while <italic>Monkey J</italic> most likely adopted a Position Control strategy (<bold><xref rid="fig8" ref-type="fig">Figure 8C</xref></bold>).</p>
<fig id="fig8" position="float" orientation="portrait" fig-type="figure">
<label>Figure 8:</label>
<caption><title>Inferring control strategies in monkeys and humans who received no instructions.</title>
<p><bold>A</bold>. Simulated data in the RMS space of cursor movement was used as training set for a classifier to determine the control objective of a trial without explicit instructions. <bold>B</bold>. Data from three example human subjects with no instructions (NI) about the control objective. Each trial (data point) is classified based on the probability of position control, P(pos), obtained for each trial from the classifier. Trials with P(pos)&gt;70% and P(pos)&lt;30% were, respectively, labeled as position control (brown) and velocity control (cyan), while other probabilities were labeled as uncertain (grey). Two example trials, one from each control objective, are shown in the bottom row. <bold>C</bold>. The classifier was used on data from two monkeys (Monkey I and J) who performed the CST. Similarly, trials for each monkey were categorized as position control (brown), velocity control (cyan), or uncertain (grey). <bold>D</bold>. Overall probability of an individual preferring the position control strategy, shown for six humans and two monkeys. This measure was obtained for each individual as the average probability of position control across all trials.</p></caption>
<graphic xlink:href="539055v1_fig8.tif" mime-subtype="tiff" mimetype="image"/>
</fig>
<p>Ultimately, our procedure enabled us to not only infer the underlying control strategy at a single trial level, but also identify which control strategy was overall preferred by humans and monkeys when no explicit knowledge about their control strategy was available. These results are encouraging as they constitute an important step towards bridging our findings between human and monkey research, and ultimately guide neurophysiological analyses to identify the neural underpinnings of control strategy in the primates’ brain.</p>
</sec>
</sec>
</sec>
<sec id="s3">
<title>Discussion</title>
<p>As we seek to understand the neural basis of human motor control, it is important to build links between studies in humans, where behavior can be complex and naturalistic, and monkeys, where direct neural recordings are possible. Doing so requires close coordination between researchers who work with humans and animals(<xref ref-type="bibr" rid="c1">Badre et al., 2015</xref>). With the goal to advance insights into movement control, the current work developed a novel approach to parallel human-monkey behavior. In a matching task design humans and monkeys performed a virtual balancing task, where they controlled an unstable system using lateral movements of their right hand to keep a cursor on the screen. The task was challenging and, importantly, exhibited different ways to achieve task success. The task required skill, but that was conceptually simple enough for monkeys to learn the skill and ultimately achieve the same level of proficiency as humans.</p>
<p>The results showed that both humans and monkeys exhibited the same behavioral characteristics as the task was made progressively more difficult: success rates dropped in a sigmoidal fashion, the correlation magnitude between hand and cursor increased, and the response lag from cursor movement to hand response decreased. Further observations based on single trials showed that the task was possibly achieved with different control strategies, both across subjects and across trials. Our goal was to identify the underlying control objectives that led to different behavior, a model based on optimal feedback control was developed that identified two different control objectives that successfully captured the average performance features of humans and monkeys: Position Control and Velocity Control. Both strategies produced behavior that was consistent with observations even at the single trial level. Additional experiments revealed that humans who followed specific instructions as to performing the task with Position Control (“keep the cursor at the center”) or Velocity Control (“keep the cursor still”) matched the behavior predicted by the two simulated control policies. Model simulations exhibited features that served to identify control strategies of humans and monkeys who received no specific instructions at a single trial level.</p>
<p>Studies in motor neurophysiology have largely relied on simple paradigms such as center-out movements (<xref ref-type="bibr" rid="c2">Batista et al., 1999</xref>; <xref ref-type="bibr" rid="c5">Cisek et al., 2003</xref>; <xref ref-type="bibr" rid="c12">Georgopoulos et al., 1986</xref>; <xref ref-type="bibr" rid="c27">Pruszynski et al., 2011</xref>; <xref ref-type="bibr" rid="c35">Scott &amp; Kalaska, 1997</xref>), which were brief in duration, highly stereotypical across repetitions, and could be performed to a reasonable degree of success with limited sensory feedback. Such characteristics were needed to make sense of noisy neural data through averaging trials over repeats of highly similar behaviors. However, such tasks are not common in natural settings, where we continually utilize sensory feedback to respond to our environment, interact with objects around us, and never do the same action the exact same way. Indeed, such fluid, prolonged and feedback-driven interactions are what we seek to understand both at the behavioral and neural levels. To this end, we need to investigate more complex tasks that involve sensory-driven control and allow for different control strategies while still within a sufficiently controlled scope. The task employed here, the Critical Stability Task (CST) offers advantages for the study of sensorimotor control that complements previously used tasks. The task continuously engages feedback-driven control mechanisms for a prolonged period of time and is rich in its trial-to-trial and subject-to-subject variability. As we can titrate the difficulty of the task, both monkeys and humans can learn it and we can study and model their behavior. This opens the gate towards understanding the neural principles of skill learning beyond simple reaching tasks. This study showed that CST afforded the examination of control strategies through a computational approach that modelled monkey and human behavior in comparable fashion.</p>
<p>A critical step for bridging insights between human and monkey behavior is through the computational approach that could explain behavior equally well in both human and monkey performance (<xref ref-type="bibr" rid="c1">Badre et al., 2015</xref>; <xref ref-type="bibr" rid="c31">Rajalingham et al., 2022</xref>). In an earlier attempt of modeling CST, a simple PD controller with delay in sensory feedback was proposed to explain the recorded behavior (<xref ref-type="bibr" rid="c30">Quick et al., 2018</xref>). However, the model was limited in its ability to capture most features observed in the data, such as success rate, or correlation between hand and cursor movements. In the past years, Optimal Feedback Control (OFC) has been introduced as an effective approach to understanding the control mechanisms of reaching movements at the level of behavior (<xref ref-type="bibr" rid="c11">Diedrichsen et al., 2010</xref>; <xref ref-type="bibr" rid="c20">McNamee &amp; Wolpert, 2019</xref>; <xref ref-type="bibr" rid="c28">Pruszynski &amp; Scott, 2012</xref>; <xref ref-type="bibr" rid="c34">Scott, 2004</xref>; <xref ref-type="bibr" rid="c38">Todorov, 2004</xref>), separately in human research (<xref ref-type="bibr" rid="c19">Liu &amp; Todorov, 2007</xref>; <xref ref-type="bibr" rid="c23">Nagengast et al., 2010</xref>; <xref ref-type="bibr" rid="c24">Nashed et al., 2014</xref>; <xref ref-type="bibr" rid="c32">Razavian et al., 2023</xref>; <xref ref-type="bibr" rid="c33">Ronsse et al., 2010</xref>; <xref ref-type="bibr" rid="c39">Todorov, 2005</xref>; <xref ref-type="bibr" rid="c40">Todorov &amp; Jordan, 2002</xref>; <xref ref-type="bibr" rid="c43">Yeo et al., 2016</xref>) and monkey research (<xref ref-type="bibr" rid="c3">Benyamini &amp; Zacksenhouse, 2015</xref>; <xref ref-type="bibr" rid="c7">Cross et al., 2023</xref>; <xref ref-type="bibr" rid="c16">Kalidindi et al., 2021</xref>; <xref ref-type="bibr" rid="c17">Kao et al., 2021</xref>; <xref ref-type="bibr" rid="c37">Takei et al., 2021</xref>). Here, the OFC framework was used to account for and make novel predictions about behavioral features in CST. Note that there are fundamental differences between reaching and CST movements, which needed to be accounted for in the modeling process. Unlike center-out reaching, the CST did not have a stationary target toward which the hand needed to move; rather, it required the hand/cursor to remain anywhere within a predefined area for a prolonged period of time. Also, the behavior was not tracking a point on the screen, but rather moving in opposite direction of the cursor, a behavior that probably requires more cognitive resources. Despite these advanced task features, OFC as a feedback control framework proved an appropriate approach to examine this demanding interactive and sensory-driven task.</p>
<p>Two aspects in our computational approach are worth discussing. First, we examined control strategies that only involved two main kinematic quantities of movement: cursor position and cursor velocity. One might argue that other kinematic features could be explored, such as acceleration or other higher derivatives of the cursor and/or the hand. However, it is important to note that, given the task of keeping the cursor within a specified area for a period of time, cursor position and velocity are the most directly related quantities to the goal of the task. These quantities were also less demanding to predict from sensory feedback, compared to, for example, acceleration (<xref ref-type="bibr" rid="c13">Hwang et al., 2006</xref>; <xref ref-type="bibr" rid="c36">Sing et al., 2009</xref>). Also note that the kinematics of the hand were not the variables of interest in the task, as the goal was to control the cursor, and not the hand.</p>
<p>Second, we mainly explored Position and Velocity Control strategies separately to identify distinctive behavioral features associated with each one. Experimental data, however, shows that a large number of trials fall somewhere between the Position and Velocity Control boundaries (<xref rid="fig7" ref-type="fig">Figure 7</xref> and <xref rid="fig8" ref-type="fig">8</xref>). This could be due to a mixed control strategy, where both Position and Velocity control strategies contribute simultaneously to achieving the task goal, or where subjects switch strategies of their own accord. Here, we aimed to determine the behavioral signatures of the extreme cases, either predominantly based on position, or velocity of the cursor movement. This may increase the chance to detect differences more clearly in neural activity associated with each control objective in further analysis of monkeys’ neurophysiological data. Even though in this experiment only a subset of trials is amenable to a clear identification as using one control strategy or another, with monkeys it is possible to collect tens of thousands of trials over many days accumulating enough trials for analysis.</p>
<p>Despite potential limitations, our approach was successful in two main ways. First, it provided a normative explanation for the macro-level characteristics of behavior observed in human and monkey data. Second, due to its generative nature, model simulations provided for not yet seen conditions and made predictions about the behavior under new control objectives. In the future, our behavioral analysis can serve as a foundation to classify or parse neural activity in monkeys performing complex actions where trial averaging is no longer possible. This behavioral analysis holds promise to generate crucial insights into neural principles of skillful manipulation, not only in monkeys but also, by induction, in humans.</p>
</sec>
<sec id="s4">
<title>Methods</title>
<sec id="s4a">
<title>Participants and Ethics Statement</title>
<p>18 healthy, right-handed university students (age: 18—25 years; 8 females) with no self-reported neuromuscular pathology volunteered to take part in the experiments. All participants were naïve to the purpose of the experiment and provided informed written consent prior to participation. The experimental paradigm and procedure were approved by the Northeastern University Institutional Review Board (IRB# 22-02-15).</p>
<p>The data from two adult male Rhesus monkeys (Macaca mulatta) used in this study was taken from a previously published work (<xref ref-type="bibr" rid="c30">Quick et al., 2018</xref>). All animal procedures were approved by the University of Pittsburgh Institutional Animal Care and Use Committee, in accordance with the guidelines of the US Department of Agriculture, the International Association for the Assessment and Accreditation of Laboratory Animal Care, and the National Institutes of Health. For details of experimental rig and procedure see the Methods in (<xref ref-type="bibr" rid="c30">Quick et al., 2018</xref>).</p>
</sec>
<sec id="s4b">
<title>Critical Stability Task (CST)</title>
<p>The CST involved balancing an unstable cursor displayed on the screen using the movement of the hand (<xref ref-type="bibr" rid="c15">Jex et al., 1966</xref>; <xref ref-type="bibr" rid="c29">Quick et al., 2014</xref>, <xref ref-type="bibr" rid="c30">2018</xref>). The CST dynamics was governed by a first-order differential equation as shown in <xref ref-type="disp-formula" rid="eqn1">equation (1)</xref>. The difficulty of the task was manipulated by changing the parameter <italic>λ</italic>: by increasing <italic>λ</italic> the task became more unstable, hence more difficult to accomplish. To perform the task, subjects sat on a sturdy chair behind a small table, with their right hand free to move above the table (<bold><xref rid="fig1" ref-type="fig">Figure 1</xref></bold>). A reflective marker was attached to the subject’s back of the hand on the third metacarpal, and the hand position was recorded using a 12-camera motion capture system at a sampling rate of 250Hz (Qualisys, 5+, Goetheburg, SE). The mediolateral component of the hand position was used to solve the CST dynamics with the initial condition of <italic>x</italic>(<italic>t</italic> = 0) = 0 (<xref ref-type="bibr" rid="c30">Quick et al., 2018</xref>). The calculated cursor position was real-time projected as a small blue disk (diameter: 4mm, approximately 0.8deg in visual angle) on a large vertical screen in front of the subject at a 150cm distance. The processing delay of the visual rendering was roughly 50ms.</p>
</sec>
<sec id="s4c">
<title>Experimental Design</title>
<sec id="s4c1">
<title>Task</title>
<p>At the beginning of the experiment, human subjects held their right hand comfortably above the table and in front of their right shoulder as shown in <bold><xref rid="fig1" ref-type="fig">Figure 1</xref></bold>, where the hand position was mapped to the center of the screen. The visual display of the cursor and hand position was scaled such that the lateral hand movements of ±10cm corresponded to ±20deg of visual angle from the screen center and served as the boundaries of the workspace. Each trial started with the hand position displayed on the screen as a red cursor (diameter: 4mm, or approximately 0.8deg in visual angle). Subjects were asked to bring the red cursor to the center of the screen depicted by a small grey box (<bold><xref rid="fig1" ref-type="fig">Figure 1</xref></bold>). Once the red cursor was at the center, and after a delay of 500ms, the trial started. The red cursor disappeared and a blue cursor representing the <italic>x</italic> position in <xref ref-type="disp-formula" rid="eqn1">equation (1)</xref> appeared at the center. Subjects were instructed to keep (or ‘balance’) the blue cursor within the boundaries of the workspace for 6s for the trial to be considered successful. If the cursor escaped the workspace at any time, the trial would abort and considered as failed. Subjects were informed of the outcome of the trial by a message on the screen, reading “Well Done!” for success, and “Failed!” for failure. The next trial started after an intertrial interval of 1000ms. This feedback matched the binary reward that monkeys were given in the experiment by Quick and colleagues.</p>
</sec>
<sec id="s4c2">
<title>Experimental Paradigm and Conditions</title>
<p>Each human subject participated in the experiment for three consecutive days. At the beginning of the first day, subjects were familiarized with the experimental setup and the objectives of the task. Familiarization consisted of five CST trials with moderate difficulty level. These trials were later excluded from the analyses. The main experiment consisted of three main phases that were repeated on each day. The first and second phases of the experiment involved 15 reaction time trials and 10 tracking trials, respectively (data for reaction time and tracking trials are not reported in this study). Phase three involved the CST trials, which were performed in three blocks. In Block 1, subjects performed 30 CST trials, where the difficulty level was determined in each trial using an up-down method: starting from <italic>λ</italic> = 2.5 in the first trial, if subjects succeeded/failed on the current trial, <italic>λ</italic> was increased/decreased by Δλ = 0.2 in the next trial. By the end of Block 1, subjects had gradually converged to λ values in which the success rate was approximately 50%. This value was considered as the critical instability value (<xref ref-type="bibr" rid="c30">Quick et al., 2018</xref>), denoted by λ<sub>C</sub>, and was obtained by averaging the λ‘s of the last 5 trials of block 1.</p>
<p>In Block 2, a stepwise increase in λ was adopted: subjects started with a difficulty level of λ = 70% λ<sub>c</sub> (using λ<sub>C</sub> from the previous block). They continued until they completed 10 successful trials, or 20 trials in total (whichever occurred first). The difficulty level was then increased by Δλ = 0.2, and the procedure repeated. This incremental increase of λ continued until the subjects’ success rate for the ongoing λ dropped below 10% (i.e., less than 2 successful trials out of 20). This marked the end of the second block. In total, subjects performed approximately 120-200 trials in Block 2, depending on the individual’s performance.</p>
<p>In Block 3, subjects performed the CST under three selected difficulty levels of easy, medium, and hard, with 20 trials for each difficulty level. These levels corresponded to λ values that led to 75% success rate (easy), 50% success rate (medium) and 25% success rate (hard) obtained from each individual’s performance in Block 2. The exact values of λ<sub>75%</sub>, λ<sub>50%</sub>, and λ<sub>25%</sub> were calculated by fitting a psychometric curve to the success rate data from Block 2 as a function of λ (see <bold><xref rid="fig2" ref-type="fig">Figure 2</xref></bold>). The order of difficulty was pseudo-randomly selected for each subject. For this study, we only analyzed the CST data from Block 2 (stepwise increase in λ) as it matched the procedure used in the monkey experiment (<xref ref-type="bibr" rid="c30">Quick et al., 2018</xref>). Subjects repeated the same experimental procedure on Day 2 and 3.</p>
<p>Three groups of human subjects participated in the experiment, where each group received different instructions about the task goal. The first group was instructed to perform the CST “without failing to the best of their ability” (no-instruction group); the second group was instructed to “keep the cursor at the center of the screen at all times” (Position Control group); and the third group was instructed to “keep the cursor still anywhere within the bounds of the screen” (velocity control group).</p>
</sec>
</sec>
<sec id="s4d">
<title>Analysis</title>
<p>To evaluate the overall performance of humans and monkeys during the CST, four quantities were calculated: success rate, hand-cursor correlation, hand-cursor lag, and hand/cursor gain. For each individual, the quantities were calculated as the average across trials for each bin of λ values (bin size: 0.3, starting from λ = 1.5).</p>
<p>The success rate was obtained as the percentage of successful trials within each λ bin. A psychometric curve (a Gaussian cumulative distribution function) was then fitted to the success rate data as a function of λ to estimate λ<sub>c</sub>(critical stability, where success rate was 50%):
<disp-formula id="eqn3">
<alternatives><graphic xlink:href="539055v1_eqn3.gif" mime-subtype="gif" mimetype="image"/></alternatives>
</disp-formula>
where, ‘erf’ indicates the error function, and <italic>σ</italic> denotes the standard deviation of the Gaussian cumulative. The correlation and lag quantities (<bold><xref rid="fig2" ref-type="fig">Figure 2, B</xref></bold> and <xref rid="fig2" ref-type="fig">C</xref>) were obtained by first cross-correlating the hand and cursor trajectories in each trial, and then finding the peak correlation, and the corresponding lag (<bold><xref rid="fig2" ref-type="fig">Figure 2</xref></bold>, see also (<xref ref-type="bibr" rid="c30">Quick et al., 2018</xref>)). The hand/cursor gain (<bold><xref rid="fig2" ref-type="fig">Figure 2, D</xref></bold>) was defined as the ratio of the root mean squared (RMS) value of hand position over the RMS value of the cursor position in each trial.</p>
<p>Finally, to perform the classification analysis used in <bold><xref rid="fig7" ref-type="fig">Figure 7</xref></bold> and <bold><xref rid="fig8" ref-type="fig">Figure 8</xref></bold>, a Support Vector Machine method was applied to learn the two-class control objective labels. In order to build and train a classifier, we used ‘fitcsvm.m’ function in MATLAB, where synthetic data (RMS of cursor position and cursor velocity) was used as training set. To classify experimental data using the trained classifier, the MATLAB function ‘predict.m’ was used. Finally, the posterior probabilities over each classification (i.e., the confidence on classification) was calculated using the ‘fitPosterior.m’ function in MATLAB.</p>
</sec>
<sec id="s4e">
<title>Optimal Feedback Control Model</title>
<p>A generative model approach was used to build control agents that performed the CST with different control strategies. The model involved an optimal feedback controller that moved the hand, a point mass of <italic>m</italic>=1kg, through a simple muscle-like actuator (<xref ref-type="bibr" rid="c39">Todorov, 2005</xref>; <xref ref-type="bibr" rid="c40">Todorov &amp; Jordan, 2002</xref>).Click or tap here to enter text. The muscle model was approximated by a first-order low-pass filter that generated forces on the hand in the lateral direction as in <xref ref-type="disp-formula" rid="eqn4">equations (4)</xref> and <xref ref-type="disp-formula" rid="eqn5">(5)</xref>:
<disp-formula id="eqn4">
<alternatives><graphic xlink:href="539055v1_eqn4.gif" mime-subtype="gif" mimetype="image"/></alternatives>
</disp-formula>
<disp-formula id="eqn5">
<alternatives><graphic xlink:href="539055v1_eqn5.gif" mime-subtype="gif" mimetype="image"/></alternatives>
</disp-formula>
where <italic>F</italic> is the actuator force acting on the hand,<italic>τ</italic> is the time constant of the low-pass filter, <italic>u</italic> is the control input to the muscle, and <inline-formula><alternatives><inline-graphic xlink:href="539055v1_inline4.gif" mime-subtype="gif" mimetype="image"/></alternatives></inline-formula> is the second derivative of the hand position. These equations consist of three states: hand position <italic>p</italic>, hand velocity <inline-formula><alternatives><inline-graphic xlink:href="539055v1_inline5.gif" mime-subtype="gif" mimetype="image"/></alternatives></inline-formula> and muscle force <italic>F</italic>. Similarly, by taking first and second derivatives of <xref ref-type="disp-formula" rid="eqn1">equation (1)</xref>, three more states were added to the dynamics of the system:
<disp-formula id="eqn6">
<alternatives><graphic xlink:href="539055v1_eqn6.gif" mime-subtype="gif" mimetype="image"/></alternatives>
</disp-formula>
<disp-formula id="eqn7">
<alternatives><graphic xlink:href="539055v1_eqn7.gif" mime-subtype="gif" mimetype="image"/></alternatives>
</disp-formula></p>
<p>By combining <xref ref-type="disp-formula" rid="eqn1">equations (1)</xref> and <xref ref-type="disp-formula" rid="eqn6">(6)</xref> and then <xref ref-type="disp-formula" rid="eqn6">(6)</xref> and <xref ref-type="disp-formula" rid="eqn7">(7)</xref>, the CST equation was expanded as follows:
<disp-formula id="eqn8">
<alternatives><graphic xlink:href="539055v1_eqn8.gif" mime-subtype="gif" mimetype="image"/></alternatives>
</disp-formula></p>
<p>The advantage of the higher derivatives of CST dynamics was that it made the cursor position <italic>x</italic>, cursor velocity <inline-formula><alternatives><inline-graphic xlink:href="539055v1_inline6.gif" mime-subtype="gif" mimetype="image"/></alternatives></inline-formula> and cursor acceleration <inline-formula><alternatives><inline-graphic xlink:href="539055v1_inline7.gif" mime-subtype="gif" mimetype="image"/></alternatives></inline-formula> available to the controller. Hence, different control strategies that directly involved these states could be explored. Note that <xref ref-type="disp-formula" rid="eqn8">equation (8)</xref> required that the initial conditions of both hand and cursor position, velocity and acceleration all satisfied <xref ref-type="disp-formula" rid="eqn6">equations (6)</xref> and <xref ref-type="disp-formula" rid="eqn7">(7)</xref>. The dynamics of the system could then be captured by <xref ref-type="disp-formula" rid="eqn4">equations (4)</xref>, <xref ref-type="disp-formula" rid="eqn5">(5)</xref> and <xref ref-type="disp-formula" rid="eqn8">(8)</xref>, and represented by the state vector: <inline-formula><alternatives><inline-graphic xlink:href="539055v1_inline8.gif" mime-subtype="gif" mimetype="image"/></alternatives></inline-formula>. By adding additive signal-independent noise <italic><bold>ξ</bold><sub>t</sub></italic>, as well as multiplicative signal-dependent noise <italic>ε<sub>t</sub>C</italic>, the full dynamics of the system could be presented in state-space format:
<disp-formula id="eqn9">
<alternatives><graphic xlink:href="539055v1_eqn9.gif" mime-subtype="gif" mimetype="image"/></alternatives>
</disp-formula>
where <italic>ε<sub>t</sub></italic> and <italic><bold>ξ</bold><sub>t</sub></italic> were zero-mean Gaussian noise terms, <italic>C</italic> was the signal-dependent noise scalar, and <italic>A</italic> and <italic>B</italic> represent the dynamics of the system:
<disp-formula id="eqn10">
<alternatives><graphic xlink:href="539055v1_eqn10.gif" mime-subtype="gif" mimetype="image"/></alternatives>
</disp-formula></p>
<p>Noisy sensory feedback <italic><bold>Y</bold><sub>t</sub></italic> was given as:
<disp-formula id="eqn11">
<alternatives><graphic xlink:href="539055v1_eqn11.gif" mime-subtype="gif" mimetype="image"/></alternatives>
</disp-formula>
where <bold>ω</bold><sub><italic>t</italic></sub>was a zero-mean additive Gaussian noise, and matrix <italic>H</italic> determined the available sensory feedback from the vector of states. For our simulations, the feedback included the cursor position <italic>x</italic> and velocity <inline-formula><alternatives><inline-graphic xlink:href="539055v1_inline9.gif" mime-subtype="gif" mimetype="image"/></alternatives></inline-formula>, therefore, <italic>H</italic> was defined as: <italic>H</italic> = [1,1,0,0,0,0]. An optimal controller determined the motor command <italic>u<sub>t</sub></italic> to minimize the cost function <italic>J</italic> as follows (<xref ref-type="bibr" rid="c39">Todorov, 2005</xref>):
<disp-formula id="eqn12">
<alternatives><graphic xlink:href="539055v1_eqn12.gif" mime-subtype="gif" mimetype="image"/></alternatives>
</disp-formula>
where <italic>n</italic> was the number of time samples throughout the movement, and <italic>Q</italic> and <italic>R</italic> determined the contribution of accuracy and effort cost, respectively. In all simulations, <italic>R</italic> = 1. The matrix <italic>Q</italic>, however, was appropriately manipulated to implement different state-dependent control strategies as discussed below.</p>
<sec id="s4e1">
<title>Position Control</title>
<p>The aim of the Position Control strategy was to maintain the cursor at the center of the screen throughout the trial. This was implemented by penalizing the deviation of the cursor position <italic>x</italic> from the center. In this case, the matrix <italic>Q</italic> was set to <italic>Q</italic> = diag([<italic>q</italic>, 1, 1,1,1,1]), where <italic>q</italic> ≫ 1 was a constant. As such, the cost of deviation from the center for the cursor position was dominant represented in the value <italic>J</italic> of the cost function, making the regulation of cursor position at the center the primary goal of control.</p>
</sec>
<sec id="s4e2">
<title>Velocity Control</title>
<p>The Velocity Control strategy aimed to keep the cursor still at any point within the boundaries of the workspace. In this case, upon deviation of the cursor from the center, the main goal was to bring the cursor to a stop regardless of the location. This was implemented through penalizing the cursor velocity <inline-formula><alternatives><inline-graphic xlink:href="539055v1_inline10.gif" mime-subtype="gif" mimetype="image"/></alternatives></inline-formula> by setting the matrix <italic>Q</italic> = diag([1, <italic>v</italic>, 1,1,1,1]), where <italic>v</italic> ≫ 1 was a constant.</p>
</sec>
<sec id="s4e3">
<title>Simulations</title>
<p>Given a control strategy, the model was used to generate 500 trials of CST for each level of task difficulty from <italic>λ</italic> = 1.5 to <italic>λ</italic> = 7, with increments of Δ<italic>λ</italic> = 0.2. The parameters of the hand and the muscle model (4)(5) were fixed to <italic>m</italic> = 1kg and<italic>τ</italic> = 0.06s. A sensory delay of 50ms was considered when simulating the task with the optimal feedback controller (<xref ref-type="bibr" rid="c39">Todorov, 2005</xref>). The signal-dependent noise terms were set to <italic>ε</italic><sub><italic>t</italic></sub>∼<italic>N</italic>(0,1), and <italic>C</italic> = 1.5. The motor noise was <italic><bold>ξ</bold><sub>t</sub></italic>∼<italic>N</italic>(<bold>0, Σ</bold>), where Σ = 0.4 <italic>BB<sup>T</sup></italic>. For each trial, the simulation started from the initial condition of <bold>x</bold> = <bold>0</bold>, and ran for 8s. Only the first 6s of each simulation were considered in the analysis for consistency with the experimental paradigm. The success or failure in each simulated trial was decided post hoc, by determining whether the cursor position <italic>x</italic> exceeded the limits of the workspace (±10cm from the center) within the 6s duration of the trial.</p>
</sec>
</sec>
</sec>
</body>
<back>
<ack>
<title>Acknowledgement</title>
<p>This research was funded by the National Institute of Health R01-CRCNS-NS120579, awarded to Dagmar Sternad and Aaron Batista. Dagmar Sternad was also supported by NIH-R37-HD087089 and NSF-M3X-1825942. Aaron Batista and Patrick Loughlin were also supported by NIH-R01-HD0909125.</p>
</ack>
<ref-list>
<title>References</title>
<ref id="c1"><label>1.</label><mixed-citation publication-type="journal"><string-name><surname>Badre</surname>, <given-names>D.</given-names></string-name>, <string-name><surname>Frank</surname>, <given-names>M. J.</given-names></string-name>, &amp; <string-name><surname>Moore</surname>, <given-names>C. I</given-names></string-name>. (<year>2015</year>). <article-title>Interactionist Neuroscience</article-title>. <source>Neuron</source>, <volume>88</volume>(<issue>5</issue>), <fpage>855</fpage>–<lpage>860</lpage>. <pub-id pub-id-type="doi">10.1016/J.NEURON.2015.10.021</pub-id></mixed-citation></ref>
<ref id="c2"><label>2.</label><mixed-citation publication-type="journal"><string-name><surname>Batista</surname>, <given-names>A. P.</given-names></string-name>, <string-name><surname>Buneo</surname>, <given-names>C. A.</given-names></string-name>, <string-name><surname>Snyder</surname>, <given-names>L. H.</given-names></string-name>, &amp; <string-name><surname>Andersen</surname>, <given-names>R. A</given-names></string-name>. (<year>1999</year>). <article-title>Reach Plans in Eye-Centered Coordinates</article-title>. <source>Science</source>, <volume>285</volume>(<issue>5425</issue>), <fpage>257</fpage>–<lpage>260</lpage>. <pub-id pub-id-type="doi">10.1126/science.285.5425.257</pub-id></mixed-citation></ref>
<ref id="c3"><label>3.</label><mixed-citation publication-type="journal"><string-name><surname>Benyamini</surname>, <given-names>M.</given-names></string-name>, &amp; <string-name><surname>Zacksenhouse</surname>, <given-names>M</given-names></string-name>. (<year>2015</year>). <article-title>Optimal feedback control successfully explains changes in neural modulations during experiments with brain-machine interfaces</article-title>. <source>Frontiers in Systems Neuroscience</source>, <volume>9</volume>. <pub-id pub-id-type="doi">10.3389/fnsys.2015.00071</pub-id></mixed-citation></ref>
<ref id="c4"><label>4.</label><mixed-citation publication-type="journal"><string-name><surname>Braun</surname>, <given-names>D. A.</given-names></string-name>, <string-name><surname>Aertsen</surname>, <given-names>A.</given-names></string-name>, <string-name><surname>Wolpert</surname>, <given-names>D. M.</given-names></string-name>, &amp; <string-name><surname>Mehring</surname>, <given-names>C</given-names></string-name>. (<year>2009</year>). <article-title>Learning optimal adaptation strategies in unpredictable motor tasks</article-title>. <source>Journal of Neuroscience</source>, <volume>29</volume>(<issue>20</issue>), <fpage>6472</fpage>–<lpage>6478</lpage>. <pub-id pub-id-type="doi">10.1523/JNEUROSCI.3075-08.2009</pub-id></mixed-citation></ref>
<ref id="c5"><label>5.</label><mixed-citation publication-type="journal"><string-name><surname>Cisek</surname>, <given-names>P.</given-names></string-name>, <string-name><surname>Crammond</surname>, <given-names>D. J.</given-names></string-name>, &amp; <string-name><surname>Kalaska</surname>, <given-names>J. F</given-names></string-name>. (<year>2003</year>). <article-title>Neural Activity in Primary Motor and Dorsal Premotor Cortex In Reaching Tasks With the Contralateral Versus Ipsilateral Arm</article-title>. <source>Journal of Neurophysiology</source>, <volume>89</volume>(<issue>2</issue>), <fpage>922</fpage>–<lpage>942</lpage>. <pub-id pub-id-type="doi">10.1152/jn.00607.2002</pub-id></mixed-citation></ref>
<ref id="c6"><label>6.</label><mixed-citation publication-type="journal"><string-name><surname>Crevecoeur</surname>, <given-names>F.</given-names></string-name>, <string-name><surname>Scott</surname>, <given-names>S. H.</given-names></string-name>, &amp; <string-name><surname>Cluff</surname>, <given-names>T</given-names></string-name>. (<year>2019</year>). <article-title>Robust Control in Human Reaching Movements: A Model-Free Strategy to Compensate for Unpredictable Disturbances</article-title>. <source>Journal of Neuroscience</source>, <volume>39</volume>(<issue>41</issue>), <fpage>8135</fpage>–<lpage>8148</lpage>. <pub-id pub-id-type="doi">10.1523/JNEUROSCI.0770-19.2019</pub-id></mixed-citation></ref>
<ref id="c7"><label>7.</label><mixed-citation publication-type="journal"><string-name><surname>Cross</surname>, <given-names>K. P.</given-names></string-name>, <string-name><surname>Guang</surname>, <given-names>H.</given-names></string-name>, &amp; <string-name><surname>Scott</surname>, <given-names>S. H</given-names></string-name>. (<year>2023</year>). <article-title>Proprioceptive and Visual Feedback Responses in Macaques Exploit Goal Redundancy</article-title>. <source>Journal of Neuroscience</source>, <volume>43</volume>(<issue>5</issue>), <fpage>787</fpage>–<lpage>802</lpage>. <pub-id pub-id-type="doi">10.1523/JNEUROSCI.1332-22.2022</pub-id></mixed-citation></ref>
<ref id="c8"><label>8.</label><mixed-citation publication-type="journal"><string-name><surname>Croxson</surname>, <given-names>P. L.</given-names></string-name>, <string-name><surname>Walton</surname>, <given-names>M. E.</given-names></string-name>, <string-name><surname>O’Reilly</surname>, <given-names>J. X.</given-names></string-name>, <string-name><surname>Behrens</surname>, <given-names>T. E. J.</given-names></string-name>, &amp; <string-name><surname>Rushworth</surname>, <given-names>M. F. S</given-names></string-name>. (<year>2009</year>). <article-title>Effort-Based Cost-Benefit Valuation and the Human Brain</article-title>. <source>Journal of Neuroscience</source>, <volume>29</volume>(<issue>14</issue>), <fpage>4531</fpage>–<lpage>4541</lpage>. <pub-id pub-id-type="doi">10.1523/JNEUROSCI.4515-08.2009</pub-id></mixed-citation></ref>
<ref id="c9"><label>9.</label><mixed-citation publication-type="journal"><string-name><surname>Desrochers</surname>, <given-names>T. M.</given-names></string-name>, <string-name><surname>Burk</surname>, <given-names>D. C.</given-names></string-name>, <string-name><surname>Badre</surname>, <given-names>D.</given-names></string-name>, &amp; <string-name><surname>Sheinberg</surname>, <given-names>D. L</given-names></string-name>. (<year>2016</year>). <article-title>The Monitoring and Control of Task Sequences in Human and Non-Human Primates</article-title>. <source>Frontiers in Systems Neuroscience</source>, <volume>9</volume>. <pub-id pub-id-type="doi">10.3389/fnsys.2015.00185</pub-id></mixed-citation></ref>
<ref id="c10"><label>10.</label><mixed-citation publication-type="journal"><string-name><surname>Diedrichsen</surname>, <given-names>J</given-names></string-name>. (<year>2007</year>). <article-title>Optimal Task-Dependent Changes of Bimanual Feedback Control and Adaptation</article-title>. <source>Current Biology</source>, <volume>17</volume>(<issue>19</issue>), <fpage>1675</fpage>–<lpage>1679</lpage>. <pub-id pub-id-type="doi">10.1016/j.cub.2007.08.051</pub-id></mixed-citation></ref>
<ref id="c11"><label>11.</label><mixed-citation publication-type="journal"><string-name><surname>Diedrichsen</surname>, <given-names>J.</given-names></string-name>, <string-name><surname>Shadmehr</surname>, <given-names>R.</given-names></string-name>, &amp; <string-name><surname>Ivry</surname>, <given-names>R. B</given-names></string-name>. (<year>2010</year>). <article-title>The coordination of movement: optimal feedback control and beyond</article-title>. <source>Trends in Cognitive Sciences</source>, <volume>14</volume>(<issue>1</issue>), <fpage>31</fpage>–<lpage>39</lpage>. <pub-id pub-id-type="doi">10.1016/j.tics.2009.11.004</pub-id></mixed-citation></ref>
<ref id="c12"><label>12.</label><mixed-citation publication-type="journal"><string-name><surname>Georgopoulos</surname>, <given-names>A. P.</given-names></string-name>, <string-name><surname>Schwartz</surname>, <given-names>A. B.</given-names></string-name>, &amp; <string-name><surname>Kettner</surname>, <given-names>R. E</given-names></string-name>. (<year>1986</year>). <article-title>Neuronal Population Coding of Movement Direction</article-title>. <source>Science</source>, <volume>233</volume>(<issue>4771</issue>), <fpage>1416</fpage>–<lpage>1419</lpage>. <pub-id pub-id-type="doi">10.1126/science.3749885</pub-id></mixed-citation></ref>
<ref id="c13"><label>13.</label><mixed-citation publication-type="journal"><string-name><surname>Hwang</surname>, <given-names>E. J.</given-names></string-name>, <string-name><surname>Smith</surname>, <given-names>M. A.</given-names></string-name>, &amp; <string-name><surname>Shadmehr</surname>, <given-names>R</given-names></string-name>. (<year>2006</year>). <article-title>Adaptation and generalization in acceleration-dependent force fields</article-title>. <source>Experimental Brain Research</source>, <volume>169</volume>(<issue>4</issue>), <fpage>496</fpage>–<lpage>506</lpage>. <pub-id pub-id-type="doi">10.1007/s00221-005-0163-2</pub-id></mixed-citation></ref>
<ref id="c14"><label>14.</label><mixed-citation publication-type="journal"><string-name><surname>Izawa</surname>, <given-names>J.</given-names></string-name>, <string-name><surname>Rane</surname>, <given-names>T.</given-names></string-name>, <string-name><surname>Donchin</surname>, <given-names>O.</given-names></string-name>, &amp; <string-name><surname>Shadmehr</surname>, <given-names>R</given-names></string-name>. (<year>2008</year>). <article-title>Motor adaptation as a process of reoptimization</article-title>. <source>Journal of Neuroscience</source>, <volume>28</volume>(<issue>11</issue>), <fpage>2883</fpage>–<lpage>2891</lpage>. <pub-id pub-id-type="doi">10.1523/JNEUROSCI.5359-07.2008</pub-id></mixed-citation></ref>
<ref id="c15"><label>15.</label><mixed-citation publication-type="journal"><string-name><surname>Jex</surname>, <given-names>H. R.</given-names></string-name>, <string-name><surname>McDonnell</surname>, <given-names>J. D.</given-names></string-name>, &amp; <string-name><surname>Phatak</surname>, <given-names>A. v.</given-names></string-name> (<year>1966</year>). <article-title>A “Critical” Tracking Task for Manual Control Research</article-title>. <source>IEEE Transactions on Human Factors in Electronics</source>, <volume>HFE-7</volume>(<issue>4</issue>), <fpage>138</fpage>–<lpage>145</lpage>. <pub-id pub-id-type="doi">10.1109/THFE.1966.232660</pub-id></mixed-citation></ref>
<ref id="c16"><label>16.</label><mixed-citation publication-type="journal"><string-name><surname>Kalidindi</surname>, <given-names>H. T.</given-names></string-name>, <string-name><surname>Cross</surname>, <given-names>K. P.</given-names></string-name>, <string-name><surname>Lillicrap</surname>, <given-names>T. P.</given-names></string-name>, <string-name><surname>Omrani</surname>, <given-names>M.</given-names></string-name>, <string-name><surname>Falotico</surname>, <given-names>E.</given-names></string-name>, <string-name><surname>Sabes</surname>, <given-names>P. N.</given-names></string-name>, &amp; <string-name><surname>Scott</surname>, <given-names>S. H.</given-names></string-name> (<year>2021</year>). <article-title>Rotational dynamics in motor cortex are consistent with a feedback controller</article-title>. <source>ELife</source>, <volume>10</volume>. <pub-id pub-id-type="doi">10.7554/eLife.67256</pub-id></mixed-citation></ref>
<ref id="c17"><label>17.</label><mixed-citation publication-type="journal"><string-name><surname>Kao</surname>, <given-names>T.-C.</given-names></string-name>, <string-name><surname>Sadabadi</surname>, <given-names>M. S.</given-names></string-name>, &amp; <string-name><surname>Hennequin</surname>, <given-names>G</given-names></string-name>. (<year>2021</year>). <article-title>Optimal anticipatory control as a theory of motor preparation: A thalamo-cortical circuit model</article-title>. <source>Neuron</source>, <volume>109</volume>(<issue>9</issue>), <fpage>1567</fpage>–<lpage>1581</lpage>.e12. <pub-id pub-id-type="doi">10.1016/j.neuron.2021.03.009</pub-id></mixed-citation></ref>
<ref id="c18"><label>18.</label><mixed-citation publication-type="journal"><string-name><surname>Kawato</surname>, <given-names>M</given-names></string-name>. (<year>1999</year>). <article-title>Internal models for motor control and trajectory planning</article-title>. <source>Current Opinion in Neurobiology</source>, <volume>9</volume>(<issue>6</issue>), <fpage>718</fpage>–<lpage>727</lpage>. <pub-id pub-id-type="doi">10.1016/S0959-4388(99)00028-8</pub-id></mixed-citation></ref>
<ref id="c19"><label>19.</label><mixed-citation publication-type="journal"><string-name><surname>Liu</surname>, <given-names>D.</given-names></string-name>, &amp; <string-name><surname>Todorov</surname>, <given-names>E</given-names></string-name>. (<year>2007</year>). <article-title>Evidence for the flexible sensorimotor strategies predicted by optimal feedback control</article-title>. <source>Journal of Neuroscience</source>, <volume>27</volume>(<issue>35</issue>), <fpage>9354</fpage>–<lpage>9368</lpage>. <pub-id pub-id-type="doi">10.1523/JNEUROSCI.1110-06.2007</pub-id></mixed-citation></ref>
<ref id="c20"><label>20.</label><mixed-citation publication-type="journal"><string-name><surname>McNamee</surname>, <given-names>D.</given-names></string-name>, &amp; <string-name><surname>Wolpert</surname>, <given-names>D. M</given-names></string-name>. (<year>2019</year>). <article-title>Internal Models in Biological Control</article-title>. <source>Annual Review of Control, Robotics, and Autonomous Systems</source>, <volume>2</volume>(<issue>1</issue>), <fpage>339</fpage>–<lpage>364</lpage>. <pub-id pub-id-type="doi">10.1146/annurev-control-060117-105206</pub-id></mixed-citation></ref>
<ref id="c21"><label>21.</label><mixed-citation publication-type="journal"><string-name><surname>Miall</surname>, <given-names>R. C.</given-names></string-name>, <string-name><surname>Christensen</surname>, <given-names>L. O. D.</given-names></string-name>, <string-name><surname>Cain</surname>, <given-names>O.</given-names></string-name>, &amp; <string-name><surname>Stanley</surname>, <given-names>J</given-names></string-name>. (<year>2007</year>). <article-title>Disruption of State Estimation in the Human Lateral Cerebellum</article-title>. <source>PLoS Biology</source>, <volume>5</volume>(<issue>11</issue>), <fpage>e316</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pbio.0050316</pub-id></mixed-citation></ref>
<ref id="c22"><label>22.</label><mixed-citation publication-type="journal"><string-name><surname>Nagengast</surname>, <given-names>A. J.</given-names></string-name>, <string-name><surname>Braun</surname>, <given-names>D. A.</given-names></string-name>, &amp; <string-name><surname>Wolpert</surname>, <given-names>D. M</given-names></string-name>. (<year>2009</year>). <article-title>Optimal control predicts human performance on objects with internal degrees of freedom</article-title>. <source>PLoS Computational Biology</source>, <volume>5</volume>(<issue>6</issue>), <fpage>e1000419</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pcbi.1000419</pub-id></mixed-citation></ref>
<ref id="c23"><label>23.</label><mixed-citation publication-type="journal"><string-name><surname>Nagengast</surname>, <given-names>A. J.</given-names></string-name>, <string-name><surname>Braun</surname>, <given-names>D. A.</given-names></string-name>, &amp; <string-name><surname>Wolpert</surname>, <given-names>D. M</given-names></string-name>. (<year>2010</year>). <article-title>Risk-Sensitive Optimal Feedback Control Accounts for Sensorimotor Behavior under Uncertainty</article-title>. <source>PLoS Computational Biology</source>, <volume>6</volume>(<issue>7</issue>), <fpage>e1000857</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pcbi.1000857</pub-id></mixed-citation></ref>
<ref id="c24"><label>24.</label><mixed-citation publication-type="journal"><string-name><surname>Nashed</surname>, <given-names>J. Y.</given-names></string-name>, <string-name><surname>Crevecoeur</surname>, <given-names>F.</given-names></string-name>, &amp; <string-name><surname>Scott</surname>, <given-names>S. H</given-names></string-name>. (<year>2014</year>). <article-title>Rapid Online Selection between Multiple Motor Plans</article-title>. <source>Journal of Neuroscience</source>, <volume>34</volume>(<issue>5</issue>), <fpage>1769</fpage>–<lpage>1780</lpage>. <pub-id pub-id-type="doi">10.1523/JNEUROSCI.3063-13.2014</pub-id></mixed-citation></ref>
<ref id="c25"><label>25.</label><mixed-citation publication-type="journal"><string-name><surname>Nayeem</surname>, <given-names>R.</given-names></string-name>, <string-name><surname>Bazzi</surname>, <given-names>S.</given-names></string-name>, <string-name><surname>Sadeghi</surname>, <given-names>M.</given-names></string-name>, <string-name><surname>Hogan</surname>, <given-names>N.</given-names></string-name>, &amp; <string-name><surname>Sternad</surname>, <given-names>D</given-names></string-name>. (<year>2021</year>). <article-title>Preparing to move: Setting initial conditions to simplify interactions with complex objects</article-title>. <source>PLoS Computational Biology</source>, <volume>17</volume>(<issue>12</issue>), <fpage>e1009597</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pcbi.1009597</pub-id></mixed-citation></ref>
<ref id="c26"><label>26.</label><mixed-citation publication-type="journal"><string-name><surname>Omrani</surname>, <given-names>M.</given-names></string-name>, <string-name><surname>Murnaghan</surname>, <given-names>C. D.</given-names></string-name>, <string-name><surname>Pruszynski</surname>, <given-names>J. A.</given-names></string-name>, &amp; <string-name><surname>Scott</surname>, <given-names>S. H</given-names></string-name>. (<year>2016</year>). <article-title>Distributed task-specific processing of somatosensory feedback for voluntary motor control</article-title>. <source>ELife</source>, <volume>5</volume>. <pub-id pub-id-type="doi">10.7554/eLife.13141</pub-id></mixed-citation></ref>
<ref id="c27"><label>27.</label><mixed-citation publication-type="journal"><string-name><surname>Pruszynski</surname>, <given-names>J. A.</given-names></string-name>, <string-name><surname>Kurtzer</surname>, <given-names>I.</given-names></string-name>, <string-name><surname>Nashed</surname>, <given-names>J. Y.</given-names></string-name>, <string-name><surname>Omrani</surname>, <given-names>M.</given-names></string-name>, <string-name><surname>Brouwer</surname>, <given-names>B.</given-names></string-name>, &amp; <string-name><surname>Scott</surname>, <given-names>S. H</given-names></string-name>. (<year>2011</year>). <article-title>Primary motor cortex underlies multi-joint integration for fast feedback control</article-title>. <source>Nature</source>, <volume>478</volume>(<issue>7369</issue>), <fpage>387</fpage>–<lpage>390</lpage>. <pub-id pub-id-type="doi">10.1038/nature10436</pub-id></mixed-citation></ref>
<ref id="c28"><label>28.</label><mixed-citation publication-type="journal"><string-name><surname>Pruszynski</surname>, <given-names>J. A.</given-names></string-name>, &amp; <string-name><surname>Scott</surname>, <given-names>S. H</given-names></string-name>. (<year>2012</year>). <article-title>Optimal feedback control and the long-latency stretch response</article-title>. <source>Experimental Brain Research</source>, <volume>218</volume>(<issue>3</issue>), <fpage>341</fpage>–<lpage>359</lpage>. <pub-id pub-id-type="doi">10.1007/s00221-012-3041-8</pub-id></mixed-citation></ref>
<ref id="c29"><label>29.</label><mixed-citation publication-type="confproc"><string-name><surname>Quick</surname>, <given-names>K. M.</given-names></string-name>, <string-name><surname>Card</surname>, <given-names>N. S.</given-names></string-name>, <string-name><surname>Whaite</surname>, <given-names>S. M.</given-names></string-name>, <string-name><surname>Mischel</surname>, <given-names>J.</given-names></string-name>, <string-name><surname>Loughlin</surname>, <given-names>P.</given-names></string-name>, &amp; <string-name><surname>Batista</surname>, <given-names>A. P</given-names></string-name>. (<year>2014</year>). <article-title>Assessing vibrotactile feedback strategies by controlling a cursor with unstable dynamics</article-title>. <conf-name>2014 36th Annual International Conference of the IEEE Engineering in Medicine and Biology Society</conf-name>, <fpage>2589</fpage>–<lpage>2592</lpage>. <pub-id pub-id-type="doi">10.1109/EMBC.2014.6944152</pub-id></mixed-citation></ref>
<ref id="c30"><label>30.</label><mixed-citation publication-type="journal"><string-name><surname>Quick</surname>, <given-names>K. M.</given-names></string-name>, <string-name><surname>Mischel</surname>, <given-names>J. L.</given-names></string-name>, <string-name><surname>Loughlin</surname>, <given-names>P. J.</given-names></string-name>, &amp; <string-name><surname>Batista</surname>, <given-names>A. P</given-names></string-name>. (<year>2018</year>). <article-title>The critical stability task: quantifying sensory-motor control during ongoing movement in nonhuman primates</article-title>. <source>Journal of Neurophysiology</source>, <volume>120</volume>(<issue>5</issue>), <fpage>2164</fpage>–<lpage>2181</lpage>. <pub-id pub-id-type="doi">10.1152/jn.00300.2017</pub-id></mixed-citation></ref>
<ref id="c31"><label>31.</label><mixed-citation publication-type="journal"><string-name><surname>Rajalingham</surname>, <given-names>R.</given-names></string-name>, <string-name><surname>Piccato</surname>, <given-names>A.</given-names></string-name>, &amp; <string-name><surname>Jazayeri</surname>, <given-names>M</given-names></string-name>. (<year>2022</year>). <article-title>Recurrent neural networks with explicit representation of dynamic latent variables can mimic behavioral patterns in a physical inference task</article-title>. <source>Nature Communications</source>, <volume>13</volume>(<issue>1</issue>), <fpage>5865</fpage>. <pub-id pub-id-type="doi">10.1038/s41467-022-33581-6</pub-id></mixed-citation></ref>
<ref id="c32"><label>32.</label><mixed-citation publication-type="other"><string-name><surname>Razavian</surname>, <given-names>R. S.</given-names></string-name>, <string-name><surname>Sadeghi</surname>, <given-names>M.</given-names></string-name>, <string-name><surname>Bazzi</surname>, <given-names>S.</given-names></string-name>, <string-name><surname>Nayeem</surname>, <given-names>R.</given-names></string-name>, &amp; <string-name><surname>Sternad</surname>, <given-names>D.</given-names></string-name> (<year>2023</year>). <article-title>Body Mechanics, Optimality, and Sensory Feedback in the Human Control of Complex Objects</article-title>. <source>Neural Computation</source>, <italic>in press</italic>, <fpage>1</fpage>–<lpage>43</lpage>. <pub-id pub-id-type="doi">10.1162/neco_a_01576</pub-id></mixed-citation></ref>
<ref id="c33"><label>33.</label><mixed-citation publication-type="journal"><string-name><surname>Ronsse</surname>, <given-names>R.</given-names></string-name>, <string-name><surname>Wei</surname>, <given-names>K.</given-names></string-name>, &amp; <string-name><surname>Sternad</surname>, <given-names>D</given-names></string-name>. (<year>2010</year>). <article-title>Optimal Control of a Hybrid Rhythmic-Discrete Task: The Bouncing Ball Revisited</article-title>. <source>Journal of Neurophysiology</source>, <volume>103</volume>(<issue>5</issue>), <fpage>2482</fpage>–<lpage>2493</lpage>. <pub-id pub-id-type="doi">10.1152/jn.00600.2009</pub-id></mixed-citation></ref>
<ref id="c34"><label>34.</label><mixed-citation publication-type="journal"><string-name><surname>Scott</surname>, <given-names>S. H</given-names></string-name>. (<year>2004</year>). <article-title>Optimal feedback control and the neural basis of volitional motor control</article-title>. <source>Nature Reviews. Neuroscience</source>, <volume>5</volume>(<issue>7</issue>), <fpage>532</fpage>–<lpage>546</lpage>. <pub-id pub-id-type="doi">10.1038/nrn1427</pub-id></mixed-citation></ref>
<ref id="c35"><label>35.</label><mixed-citation publication-type="journal"><string-name><surname>Scott</surname>, <given-names>S. H.</given-names></string-name>, &amp; <string-name><surname>Kalaska</surname>, <given-names>J. F</given-names></string-name>. (<year>1997</year>). <article-title>Reaching Movements With Similar Hand Paths But Different Arm Orientations</article-title>. <source>I. Activity of Individual Cells in Motor Cortex. Journal of Neurophysiology</source>, <volume>77</volume>(<issue>2</issue>), <fpage>826</fpage>– <lpage>852</lpage>. <pub-id pub-id-type="doi">10.1152/jn.1997.77.2.826</pub-id></mixed-citation></ref>
<ref id="c36"><label>36.</label><mixed-citation publication-type="journal"><string-name><surname>Sing</surname>, <given-names>G. C.</given-names></string-name>, <string-name><surname>Joiner</surname>, <given-names>W. M.</given-names></string-name>, <string-name><surname>Nanayakkara</surname>, <given-names>T.</given-names></string-name>, <string-name><surname>Brayanov</surname>, <given-names>J. B.</given-names></string-name>, &amp; <string-name><surname>Smith</surname>, <given-names>M. A</given-names></string-name>. (<year>2009</year>). <article-title>Primitives for motor adaptation reflect correlated neural tuning to position and velocity</article-title>. <source>Neuron</source>, <volume>64</volume>(<issue>4</issue>), <fpage>575</fpage>–<lpage>589</lpage>. <pub-id pub-id-type="doi">10.1016/j.neuron.2009.10.001</pub-id></mixed-citation></ref>
<ref id="c37"><label>37.</label><mixed-citation publication-type="journal"><string-name><surname>Takei</surname>, <given-names>T.</given-names></string-name>, <string-name><surname>Lomber</surname>, <given-names>S. G.</given-names></string-name>, <string-name><surname>Cook</surname>, <given-names>D. J.</given-names></string-name>, &amp; <string-name><surname>Scott</surname>, <given-names>S. H</given-names></string-name>. (<year>2021</year>). <article-title>Transient deactivation of dorsal premotor cortex or parietal area 5 impairs feedback control of the limb in macaques</article-title>. <source>Current Biology</source>, <volume>31</volume>(<issue>7</issue>), <fpage>1476</fpage>–<lpage>1487</lpage>.e5. <pub-id pub-id-type="doi">10.1016/j.cub.2021.01.049</pub-id></mixed-citation></ref>
<ref id="c38"><label>38.</label><mixed-citation publication-type="journal"><string-name><surname>Todorov</surname>, <given-names>E</given-names></string-name>. (<year>2004</year>). <article-title>Optimality principles in sensorimotor control</article-title>. <source>Nature Neuroscience</source>, <volume>7</volume>(<issue>9</issue>), <fpage>907</fpage>–<lpage>915</lpage>. <pub-id pub-id-type="doi">10.1038/nn1309</pub-id></mixed-citation></ref>
<ref id="c39"><label>39.</label><mixed-citation publication-type="journal"><string-name><surname>Todorov</surname>, <given-names>E</given-names></string-name>. (<year>2005</year>). <article-title>Stochastic optimal control and estimation methods adapted to the noise characteristics of the sensorimotor system</article-title>. <source>Neural Computation</source>, <volume>17</volume>(<issue>5</issue>), <fpage>1084</fpage>–<lpage>1108</lpage>. <pub-id pub-id-type="doi">10.1162/0899766053491887</pub-id></mixed-citation></ref>
<ref id="c40"><label>40.</label><mixed-citation publication-type="journal"><string-name><surname>Todorov</surname>, <given-names>E.</given-names></string-name>, &amp; <string-name><surname>Jordan</surname>, <given-names>M. I</given-names></string-name>. (<year>2002</year>). <article-title>Optimal feedback control as a theory of motor coordination</article-title>. <source>Nature Neuroscience</source>, <volume>5</volume>(<issue>11</issue>), <fpage>1226</fpage>–<lpage>1235</lpage>. <pub-id pub-id-type="doi">10.1038/nn963</pub-id></mixed-citation></ref>
<ref id="c41"><label>41.</label><mixed-citation publication-type="journal"><string-name><surname>Uno</surname>, <given-names>Y.</given-names></string-name>, <string-name><surname>Kawato</surname>, <given-names>M.</given-names></string-name>, &amp; <string-name><surname>Suzuki</surname>, <given-names>R</given-names></string-name>. (<year>1989</year>). <article-title>Formation and control of optimal trajectory in human multijoint arm movement</article-title>. <source>Biological Cybernetics</source>, <volume>61</volume>(<issue>2</issue>), <fpage>89</fpage>–<lpage>101</lpage>. <pub-id pub-id-type="doi">10.1007/BF00204593</pub-id></mixed-citation></ref>
<ref id="c42"><label>42.</label><mixed-citation publication-type="journal"><string-name><surname>Wong</surname>, <given-names>J. D.</given-names></string-name>, <string-name><surname>Cluff</surname>, <given-names>T.</given-names></string-name>, &amp; <string-name><surname>Kuo</surname>, <given-names>A. D</given-names></string-name>. (<year>2021</year>). <article-title>The energetic basis for smooth human arm movements</article-title>. <source>ELife</source>, <volume>10</volume>. <pub-id pub-id-type="doi">10.7554/eLife.68013</pub-id></mixed-citation></ref>
<ref id="c43"><label>43.</label><mixed-citation publication-type="journal"><string-name><surname>Yeo</surname>, <given-names>S.-H.</given-names></string-name>, <string-name><surname>Franklin</surname>, <given-names>D. W.</given-names></string-name>, &amp; <string-name><surname>Wolpert</surname>, <given-names>D. M</given-names></string-name>. (<year>2016</year>). <article-title>When Optimal Feedback Control Is Not Enough: Feedforward Strategies Are Required for Optimal Control with Active Sensing</article-title>. <source>PLoS Computational Biology</source>, <volume>12</volume>(<issue>12</issue>), <fpage>e1005190</fpage>. <pub-id pub-id-type="doi">10.1371/journal.pcbi.1005190</pub-id></mixed-citation></ref>
</ref-list>
</back>
<sub-article id="sa0" article-type="editor-report">
<front-stub>
<article-id pub-id-type="doi">10.7554/eLife.88514.1.sa3</article-id>
<title-group>
<article-title>eLife Assessment</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Cowan</surname>
<given-names>Noah J</given-names>
</name>
<role specific-use="editor">Reviewing Editor</role>
<aff>
<institution-wrap>
<institution>Johns Hopkins University</institution>
</institution-wrap>
<city>Baltimore</city>
<country>United States of America</country>
</aff>
</contrib>
</contrib-group>
<kwd-group kwd-group-type="evidence-strength">
<kwd>Incomplete</kwd>
</kwd-group>
<kwd-group kwd-group-type="claim-importance">
<kwd>Valuable</kwd>
</kwd-group>
</front-stub>
<body>
<p>This study has the potential to provide <bold>valuable</bold> insights that connect the way humans and non-human primates (Rhesus monkeys) perform visuomotor control for a simplified, virtual task that involves stabilizing an unstable system (analogous to pole balancing). Thus, the paper provides the potential, in future studies, to make new discoveries in the neural mechanisms that underly behavior. However, the evidence (including inferring control strategies on a single-trial basis) was <bold>incomplete</bold>. Overall, the question and approach are potentially <bold>valuable</bold> in informing future studies, but more evidence is needed to support the primary claims of the paper.</p>
</body>
</sub-article>
<sub-article id="sa1" article-type="referee-report">
<front-stub>
<article-id pub-id-type="doi">10.7554/eLife.88514.1.sa2</article-id>
<title-group>
<article-title>Reviewer #1 (Public Review):</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<anonymous/>
<role specific-use="referee">Reviewer</role>
</contrib>
</contrib-group>
</front-stub>
<body>
<p>The present study examines whether one can identify kinematic signatures of different motor strategies in both humans and non-human primates (NHP). The Critical Stability Task (CST) requires a participant to control a cursor with complex dynamics based on hand motion. The manuscript includes datasets on performance of NHPs collected from a previous study, as well as new data on humans performing the same task. Further human experiments and optimal control models highlight how different strategies lead to different patterns of hand motion. Finally, classifiers were developed to predict which strategy individuals were using on a given trial. There are several strengths to this manuscript. I think the CST task provides a useful behavioural task to explore the neural basis of voluntary control. While reaching is an important basic motor skill, there is much to learn by looking at other motor actions to address many fundamental issues on the neural basis of voluntary control. I also think the comparison between human and NHP performance is important as there is a common concern that NHPs can be overtrained in performing motor tasks leading to differences in their performance as compared to humans. The present study highlights that there are clear similarities in motor strategies of humans and NHPs. While the results are promising, I would suggest that the actual use of these paradigms and techniques likely need some improvement/refinement. Notably, the threshold or technique to identify which strategy an individual is using on a given trial needs to be more stringent given the substantial overlap in hand kinematics between different strategies.</p>
<p>The most important goal of this study is to set up future studies to examine how changes in motor strategies impact neural processing. I have a few concerns that I think need to be considered. First, a classifier was developed to identify whether a trial reflected Position Control with success deemed to be a probability of &gt;70% by the classifier. In contrast, a probability of &lt;30% was considered successfully predicting Velocity Control (Uncertain bandwidth middle 40%). While this may be viewed as acceptable for purposes of quantifying behaviour, I'm not sure this is strict enough for interpreting neural data. Figure 7A displays the OFC Model results for the two strategies and demonstrates substantial overlap for RMS of Cursory Position and Velocity at the lowest range of values. In this region, individual trials for humans and NHP are commonly identified as reflecting Position Control by the classifier although this region clearly also falls within the range expected for Velocity Control, just a lower density of trials. The problem is that neural data is messy enough, but having trials being incorrectly labelled will make it even messier when trying to quantify differences in neural processing between strategies. A further challenge is that trials cannot be averaged as the patterns of kinematics are so different from trial-to-trial. One option is to just move up the threshold from &gt;70%/&lt;30% to levels where you have a higher confidence that performance only reflects one of the two strategies (perhaps 95/5% level). Another approach would be to identify the 95% confidence boundary for a given strategy and only classify a trial as reflecting a given strategy when it is inside its 95% boundary, but outside the other strategies 95% boundary (or some other level separation). A higher threshold would hopefully also deal with the challenge of individuals switching strategies within a trial. Admittedly, this more stringent separation will likely drop the number of trials prohibitively, but there is a clear trade-off between number of trials and clean data. For the future, a tweak to the task could be to lengthen the trial as this would certainly increase separation between the two conditions.</p>
<p>While the paradigm creates interesting behavioural differences, it is not clear to me what one would expect to observe neurally in different brain regions beyond paralleling kinematic differences in performance. Perhaps this could be discussed. One extension of the present task would be to add some trials where visual disturbances are applied near the end of the trial. The prediction is that there would be differences in the kinematics of these motor corrections for different motor strategies. One could then explore differences in neural processing across brain regions to identify regions that simply reflect sensory feedback (no differences in the neural response after the disturbance), versus those involved in different motor strategies (differences in neural responses after the disturbance).</p>
<p>It seems like a mix of lambda values are presented in Figure 5 and beyond. There needs to be some sort of analysis to verify that all strategies were equally used across lambda levels. Otherwise, apparent differences between control strategies may simply reflect changes in the difficulty of the task. It would also be useful to know if there were any trends across time? Strategies used for blocks of trials or one used early when learning and then changing later.</p>
<p>Figure 2 highlights key features of performance as a function of task difficulty. Lines 187 to 191 highlight similarities in motor performance between humans and NHPs. However, there is a curious difference in hand/cursor Gain for Monkey J. Any insight as to the basis for this difference?</p>
</body>
</sub-article>
<sub-article id="sa2" article-type="referee-report">
<front-stub>
<article-id pub-id-type="doi">10.7554/eLife.88514.1.sa1</article-id>
<title-group>
<article-title>Reviewer #2 (Public Review):</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<anonymous/>
<role specific-use="referee">Reviewer</role>
</contrib>
</contrib-group>
</front-stub>
<body>
<p>The goal of the present study is to better understand the 'control objectives' that subjects adopt in a video-game-like virtual-balancing task. In this task, the hand must move in the opposite direction from a cursor. For example, if the cursor is 2 cm to the right, the subject must move their hand 2 cm to the left to 'balance' the cursor. Any imperfection in that opposition causes the cursor to move. E.g., if the subject were to move only 1.8 cm, that would be insufficient, and the cursor would continue to move to the right. If they were to move 2.2 cm, the cursor would move back toward the center of the screen. This return to center might actually be 'good' from the subject's perspective, depending on whether their objective is to keep the cursor still or keep it near the screen's center. Both are reasonable 'objectives' because the trial fails if the cursor moves too far from the screen's center during each six-second trial.</p>
<p>This task was recently developed for use in monkeys (Quick et al., 2018), with the intention of being used for the study of the cortical control of movement, and also as a task that might be used to evaluate BMI control algorithms. The purpose of the present study is to better characterize how this task is performed. What sort of control policies are used. Perhaps more deeply, what kind of errors are those policies trying to minimize? To address these questions, the authors simulate control-theory style models and compare with behavior. They do in both in monkeys and in humans.</p>
<p>These goals make sense as a precursor to future recording or BMI experiments. The primate motor-control field has long been dominated by variants of reaching tasks, so introducing this new task will likely be beneficial. This is not the first non-reaching task, but it is an interesting one and it makes sense to expand the presently limited repertoire of tasks. The present task is very different from any prior task I know of. Thus, it makes sense to quantify behavior as thoroughly as possible in advance of recordings. Understanding how behavior is controlled is, as the authors note, likely to be critical to interpreting neural data.</p>
<p>From this perspective - providing a basis for interpreting future neural results - the present study is fairly successful. Monkeys seem to understand the task properly, and to use control policies that are not dissimilar from humans. Also reassuring is the fact that behavior remains sensible even when task-difficulty become high. By 'sensible' I simply mean that behavior can be understood as seeking to minimize error: position, velocity, or (possibly) both, and that this remains true across a broad range of task difficulties. The authors document why minimizing position and minimizing velocity are both reasonable objectives. Minimizing velocity is reasonable, because a near-stationary cursor can't move far in six seconds. Minimizing position error is reasonable, because the trial won't fail if the cursor doesn't stray far from the center. This is formally demonstrated by simulating control policies: both objectives lead to control policies that can perform the task and produce realistic single-trial behavior. The authors also demonstrate that, via verbal instruction, they can induce human subjects to favor one objective over the other. These all seem like things that are on the 'need to know' list, and it is commendable that this amount of care is being taken before recordings begin, as it will surely aid interpretation.</p>
<p>Yet as a stand-alone study, the contribution to our understanding of motor control is more limited. The task allows two different objectives (minimize velocity, minimize position) to be equally compatible with the overall goal (don't fail the trial). Or more precisely, there exists a range of objectives with those two at the extreme. So it makes sense that different subjects might choose to favor different objectives, and also that they can do so when instructed. But has this taught us something about motor control, or simply that there is a natural ambiguity built into the task? If I ask you to play a game, but don't fully specify the rules, should I be surprised that different people think the rules are slightly different?</p>
<p>The most interesting scientific claim of this study is not the subject-to-subject variability; the task design makes that quite likely and natural. Rather, the central scientific result is the claim that individual subjects are constantly switching objectives (and thus control policies), such that the policy guiding behavior differs dramatically even on a single-trial basis. This scientific claim is supported by a technical claim: that the authors' methods can distinguish which objective is in use, even on single trials. I am uncertain of both claims.</p>
<p>Consider Figure 8B, which reprises a point made in Figure 1&amp;3 and gives the best evidence for trial-to-trial variability in objective/policy. For every subject, there are two example trials. The top row of trials shows oscillations around the center, which could be consistent with position-error minimization. The bottom row shows tolerance of position errors so long as drift is slow, which could be consistent with velocity-error minimization. But is this really evidence that subjects were switching objectives (and thus control policies) from trial to trial? A simpler alternative would be a single control policy that does not switch, but still generates this range of behaviors. The authors don't really consider this possibility, and I'm not sure why. One can think of a variety of ways in which a unified policy could produce this variation, given noise and the natural instability of the system.</p>
<p>Indeed, I found that it was remarkably easy to produce a range of reasonably realistic behaviors, including the patterns that the authors interpret as evidence for switching objectives, based on a simple fixed controller. To run the simulations, I made the simple assumption that subjects simply attempt to match their hand position to oppose the cursor position. Because subjects cannot see their hand, I assumed modest variability in the gain, with a range from -1 to -1.05. I assumed a small amount of motor noise in the outgoing motor command. The resulting (very simple) controller naturally displayed the basic range of behaviors observed across trials (see Image 1)</p>
<fig id="sa2fig1">
<label>Peer review image 1.</label>
<graphic mime-subtype="jpg" xlink:href="elife-88514-sa2-fig1.jpg" mimetype="image"/>
</fig>
<p>Some trials had oscillations around the screen center (zero), which is the pattern the authors suggest reflects position control. In other trials the cursor was allowed to drift slowly away from the center, which is the pattern the authors suggest reflects velocity control. This is true even though the controller was the same on every trial. Trial-to-trial differences were driven both by motor noise and by the modest variability in gain. In an unstable system, small differences can lead to (seemingly) qualitatively different behavior on different trials.</p>
<p>This simple controller is also compatible with the ability of subjects to adapt their strategy when instructed. Anyone experienced with this task likely understands (or has learned) that moving the hand slightly more than 'one should' will tend to shepherd the cursor back to center, at the cost of briefly high velocity. Using this strategy more sparingly will tend to minimize velocity even if position errors persist. Thus, any subject using this control policy would be able to adapt their strategy via a modest change in gain (the gain linking visible cursor position to intended hand position).</p>
<p>This model is simple, and there may be reasons to dislike it. But it is presumably a reasonable model. The nature of the task is that you should move your hand opposite where the cursor is. Because you can't see your hand, you will make small mistakes. Due to the instability of the system, those small mistakes have large and variable effects. This feature is likely common to other controllers as well; many may explicitly or implicitly blend position and velocity control, with different trials appearing more dominated by one versus the other. Given this, I think the study presents only weak evidence that individual subjects are switching their objective on individual trials. Indeed, the more parsimonious explanation may be that they aren't. While the study certainly does demonstrate that the control policy can be influenced by verbal instructions, this might be a small adjustment as noted above.</p>
<p>I thus don't feel convinced that the authors can conclusively tell us the true control policy being used by human and monkey subjects, nor whether that policy is mostly fixed or constantly switching. The data are potentially compatible with any of these interpretations, depending on which control-style model one prefers.</p>
<p>I see a few paths that the authors might take if they chose.</p>
<p>
--First, my reasoning above might be faulty, or there might be additional analyses that could rule out the possibility of a unified policy underlying variable behavior. If so, the authors may be able to reject the above concerns and retain the present conclusions. The main scientifically novel conclusion of the present study is that subjects are using a highly variable control policy, and switching on individual trials. If this is indeed the case, there may be additional analyses that could reveal that.</p>
<p>
--Second, additional trial types (e.g., with various perturbations) might be used as a probe of the control policy. As noted below, there is a long history of doing this in the pursuit system. That additional data might better disambiguate control policies both in general, and across trials.</p>
<p>
--Third, the authors might find that a unified controller is actually a good (and more parsimonious) explanation. Which might actually be a good thing from the standpoint of future experiments. Interpretation of neural data is likely to be much easier if the control policy being instantiated isn't in constant flux.</p>
<p>In any case, I would recommend altering the strength of some conclusions, particularly the conclusion that the presented methods can reliably discriminate amongst objectives/policies on individual trials. This is mentioned as a major motivation on multiple occasions, but in most of these instances, the subsequent analysis infers the objective only across trial (e.g., one must observe a scatterplot of many trials). By Figure 7, they do introduce a method for inferring the control policy on individual trials, and while this seems to work considerably better than chance, it hardly appears reliable.</p>
<p>In this same vein I would suggest toning down aspects of the Introduction and Discussion. The Introduction in particular is overly long, and tries to position the present study as unique in ways that seem strained. Other studies have built links between human behavior, monkey behavior, and monkey neural data (for just one example, consider the corpus of work from the Scott lab that includes Pruszynski et al. 2008 and 2011). Other studies have used highly quantitative methods to infer the objective function used by subjects (e.g. Kording and Wolpert 2004). The very issue that is of interest in the present study - velocity-error-minimization versus position-error-minimization - has been extensively addressed in the smooth pursuit system. That field has long combined quantitative analyses of behavior in humans and monkeys, along with neural recordings. Many pursuit experiments used strategies that could be fruitfully employed to address the central questions of the present study. For example, error stabilization was important for dissecting the control policy used by the pursuit system. By artificially stabilizing the error (position or velocity) at zero, or at some other value, one can determine the system's response. The classic Rashbass step (1961) put position and velocity errors in opposition, to see which dominates the response. Step and sinusoidal perturbations were useful in distinguishing between models, as was the imposition of artificially imposed delays. The authors note the 'richness' of the behavior in the present task, and while one could say the same of pursuit, it was still the case that specific and well-thought through experimental manipulations were pretty critical. It would be better if the Introduction considered at least some of the above-mentioned work (or other work in a similar vein). While most would agree with the motivations outlined by the authors - they are logical and make sense - the present Introduction runs the risk of overselling the present conclusions while underselling prior work.</p>
</body>
</sub-article>
<sub-article id="sa3" article-type="referee-report">
<front-stub>
<article-id pub-id-type="doi">10.7554/eLife.88514.1.sa0</article-id>
<title-group>
<article-title>Reviewer #3 (Public Review):</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<anonymous/>
<role specific-use="referee">Reviewer</role>
</contrib>
</contrib-group>
</front-stub>
<body>
<p>This paper considers a challenging motor control task - the critical stability task (CST) - that can be performed equally well by humans and macaque monkeys. This task is of considerable interest since it is rich enough to potentially yield important novel insights into the neural basis of behavior in more complex tasks that point-to-point reaching. Yet it is also simple enough to allow parallel investigation in humans and monkeys, and is also easily amenable to computational modeling. The paper makes a compelling argument for the importance of this type of parallel investigation and the suitability of the CST for doing so.</p>
<p>Behavior in monkeys and in human subjects suggests that behavior seems to cluster into different regimes that seem to either oscillate about the center of the screen, or drift more slowly in one direction. The authors show that these two behavioral regimes can be reliably reproduced by instructing human participants to either maintain the cursor in the center of the screen (position control objective), or keep the cursor still anywhere in the screen (velocity control objective) - as opposed to the usual 'instruction' to just not let the cursor leave the screen. A computational model based on optimal feedback control can similarly reproduce the two control regimes when the costs are varied</p>
<p>Overall, this is a creative study that successfully leverages experiments in humans and computational modeling to gain insight into the nature of individual differences in behavior across monkeys (and people). The approach does work and successfully solves the core problem the authors set out to address. I do think that more comprehensive approaches might be possible that might involve, e.g. using a richer set of behavioral features to classify behavior, fitting a parametric class of control objectives rather than assuming a binary classification, and exploring the reliability of the inference process in more detail.</p>
<p>In addition, the authors do fully establish that varying control objectives is the only way to obtain the different behavioral phenotypes observed. It may, for instance, be possible that some other underlying differences (e.g. the sensitivity to effort costs or the extent of signal-dependent noise) might also lead to a similar range of behaviors as varying the position versus velocity costs.</p>
<p>Specific Comments:</p>
<p>
The simulations convincingly show that varying the control objective via the cost function can reproduce the different observed behavioral regimes. However, in principle, the differences in behavior among the monkeys and among the humans in Experiment 1 might not necessarily be due to difference in other aspects of the model. For instance, for a fixed cost function, differences in motor execution noise might perhaps lead the model to favor a position-like strategy or a velocity-like strategy. Or differences in the relative effort cost might alter the behavioral phenotype. Given that the narrative is about inferring control objectives, it seems important to rule out more systematically that some other factor might not potentially dictate each individual's style of performing the task. One approach to rule this out might be to try to formally fit the parameters of the model (or at least a subset of them) under a fixed cost function (e.g. velocity-based), and check whether the model might still recover the different regimes of behavior when parameters *other than the cost function* are varied.</p>
<p>The approach to the classification problem is somewhat ad hoc and based on fairly simplistic, hand-picked features (RMS position and RMS velocity). I do wonder whether a more comprehensive set of behavioral features might enable a clearer separation between strategies, or might even reveal that the uninstructed subjects were doing something qualitatively different still from the instructed groups. Different control objectives ought to predict meaningfully different control policies - that is, different ways of updating hand position based on current state of the cursor and hand - e.g. the hand/cursor gain, which does clearly differ across instructed strategies. Would it be possible to distinguish control strategies more accurately based on this level of analysis, rather than based on gross task metrics? Might this point to possible experimental interventions (e.g. target jumps) that might validate the inferred objective?</p>
<p>It seems that the classification problem cannot be solved perfectly, at least on a single-trial level. Although it works out that the classification can recover which participants were given which instructions, it's not clear how robust this classification is. It should be straightforward to estimate the reliability of the strategy classification by simulating participants and deriving a &quot;confusion matrix&quot;, i.e. calculating how often e.g. data generated under a velocity-control objective gets mis-classified as following a position-control objective. It's not clear how this kind of metric relates to the decision confidence outputted by the classifier.</p>
<p>The problem of inferring the control objective is framed as a dichotomy between position control and velocity control. In reality, however, it may be a continuum of possible objectives, based on the relative cost for position and velocity. How would the problem differ if the cost function is framed as estimating a parameter, rather than as a classification problem?</p>
</body>
</sub-article>
<sub-article id="sa4" article-type="author-comment">
<front-stub>
<article-id pub-id-type="doi">10.7554/eLife.88514.1.sa4</article-id>
<title-group>
<article-title>Author Response</article-title>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<name>
<surname>Sadeghi</surname>
<given-names>Mohsen</given-names>
</name>
<role specific-use="author">Author</role>
<contrib-id contrib-id-type="orcid">http://orcid.org/0000-0003-2573-146X</contrib-id></contrib>
<contrib contrib-type="author">
<name>
<surname>Razavian</surname>
<given-names>Reza Sharif</given-names>
</name>
<role specific-use="author">Author</role>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Bazzi</surname>
<given-names>Salah</given-names>
</name>
<role specific-use="author">Author</role>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Chowdhury</surname>
<given-names>Raeed</given-names>
</name>
<role specific-use="author">Author</role>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Batista</surname>
<given-names>Aaron</given-names>
</name>
<role specific-use="author">Author</role>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Loughlin</surname>
<given-names>Patrick</given-names>
</name>
<role specific-use="author">Author</role>
</contrib>
<contrib contrib-type="author">
<name>
<surname>Sternad</surname>
<given-names>Dagmar</given-names>
</name>
<role specific-use="author">Author</role>
</contrib>
</contrib-group>
</front-stub>
<body>
<p>We thank the reviewers for thoroughly evaluating our work and for providing constructive and actionable feedback to improve the manuscript. The reviews have left us with a clear direction in which our work can improve, for which we are grateful. We will provide a detailed response to the reviews together with our revised manuscript. At this time, we accept the invitation to provide a provisional reply that addresses the major themes as summarized by the editors.</p>
<p>The goal of our study was to infer an individual’s control strategy from the details of kinematics. We did this using monkey and human data collected under matching experimental conditions. We quantitatively compared these data to simulations that were generated by adapting a reasonable model of sensorimotor function that is standard in the literature. We are pleased that the reviewers and editors felt “that the overall scientific approach is of interest and has scientific merit” and “the approach has promise in aiding future studies that try to link behavior and neurophysiology (allowing homology between humans and primates).”</p>
<p>We agree with the reviewers that additional work is needed to corroborate our main claim that we can unambiguously infer control strategies from behavioral data. This is a known hard problem that we are not the first to address, and we do not claim to have solved it here. We appreciate the suggestions about (1) further testing the classification procedure, (2) considering other metrics that may better distinguish between the control strategies, and (3) investigating the control strategy under perturbation scenarios. We plan to undertake additional simulations, analyses and, in the future, experiments, as suggested by the reviewers to enhance the impact of our work.</p>
<p>In this initial brief response, we wish to focus on one key point noted by the editors, stemming from simulations by one of the reviewers using “a simple fixed controller.” We greatly appreciate that one reviewer went as far as to perform their own simulations. These simulations suggested that subjects do not need to switch between control strategies, but rather could achieve similar behavioral results via “a modest change in gain.” Specifically, the reviewer reports that their simple fixed controller could generate trials that sometimes looked like what we would call position control and sometimes looked like what we would characterize as velocity control. It was noted that “trial-to-trial differences were driven both by motor noise and by the modest variability in gain.”</p>
<p>While we cannot comment with great certainty on the reviewer’s simulation results, since we do not know the specifics, we first wish to note that our controller and experimental subjects demonstrated this same phenomenon, in that there was overlap in the distribution of the metrics for the two strategies (specifically, in Figs. 5, 7 &amp; 8). Hence, in our findings, even under position control some trials looked more like velocity control, and vice versa. We briefly discussed this in the paper, noting that “a large number of trials fall somewhere between the Position and Velocity Control boundaries”, and that “this could be due to a mixed control strategy” or “subjects switch strategies of their own accord”. This point would have been clearer had we included examples of these hand and cursor traces in Fig. 8. We will update Fig. 8 to more clearly illustrate this point and expand our discussion on different possible interpretations.</p>
<p>Second, one may interpret the differences we attributed to changes in “control strategy” as changes simply in the gain of our “fixed” controller. Specifically, similar to the controller implemented by the reviewer, our controller is fixed in terms of the plant, the actuator and the sensory feedback loop; the only change we explored was in the relative weights or gains of position vs. velocity in the Q matrix to generate the motor command. While our intent was primarily to focus on the extremes of position control vs. velocity control, we agree that a mixed strategy of minimizing some combined error in position and velocity is likely. This is something we can readily explore with our controller model.</p>
<p>In summary, we consider it worthwhile to investigate how one can infer the control strategy that a subject is employing to complete the task - either in our CST, or any other task that admits multiple strategies that can lead to success. We regard this as a valuable step towards addressing more realistic behaviors and their neural underpinnings in non-human primate research. The suggestions offered by the reviewers regarding additional analyses, simulations and experiments will provide more definitive answers and clarity for our approach.</p>
<p>We are truly grateful for the time and effort the reviewers put into our manuscript. We are in the process of undertaking revisions to address all of their feedback and look forward to submitting an improved manuscript with a more detailed reply in the coming weeks.</p>
</body>
</sub-article>
</article>