<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Archiving and Interchange DTD with MathML3 v1.3 20210610//EN"  "JATS-archivearticle1-3-mathml3.dtd"><article xmlns:ali="http://www.niso.org/schemas/ali/1.0/" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article" dtd-version="1.3"><front><journal-meta><journal-id journal-id-type="nlm-ta">elife</journal-id><journal-id journal-id-type="publisher-id">eLife</journal-id><journal-title-group><journal-title>eLife</journal-title></journal-title-group><issn publication-format="electronic" pub-type="epub">2050-084X</issn><publisher><publisher-name>eLife Sciences Publications, Ltd</publisher-name></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">88958</article-id><article-id pub-id-type="doi">10.7554/eLife.88958</article-id><article-id pub-id-type="doi" specific-use="version">10.7554/eLife.88958.3</article-id><article-version article-version-type="publication-state">version of record</article-version><article-categories><subj-group subj-group-type="display-channel"><subject>Research Article</subject></subj-group><subj-group subj-group-type="heading"><subject>Structural Biology and Molecular Biophysics</subject></subj-group></article-categories><title-group><article-title>Predicting the sequence-dependent backbone dynamics of intrinsically disordered proteins</article-title></title-group><contrib-group><contrib contrib-type="author"><name><surname>Qin</surname><given-names>Sanbo</given-names></name><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="fn" rid="con1"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author" corresp="yes"><name><surname>Zhou</surname><given-names>Huan-Xiang</given-names></name><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0001-9020-0302</contrib-id><email>hzhou43@uic.edu</email><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="other" rid="fund1"/><xref ref-type="fn" rid="con2"/><xref ref-type="fn" rid="conf1"/></contrib><aff id="aff1"><label>1</label><institution-wrap><institution-id institution-id-type="ror">https://ror.org/02mpq6x41</institution-id><institution>Department of Chemistry, University of Illinois Chicago</institution></institution-wrap><addr-line><named-content content-type="city">Chicago</named-content></addr-line><country>United States</country></aff><aff id="aff2"><label>2</label><institution-wrap><institution-id institution-id-type="ror">https://ror.org/02mpq6x41</institution-id><institution>Department of Physics, University of Illinois Chicago</institution></institution-wrap><addr-line><named-content content-type="city">Chicago</named-content></addr-line><country>United States</country></aff></contrib-group><contrib-group content-type="section"><contrib contrib-type="editor"><name><surname>Cui</surname><given-names>Qiang</given-names></name><role>Reviewing Editor</role><aff><institution-wrap><institution-id institution-id-type="ror">https://ror.org/05qwgg493</institution-id><institution>Boston University</institution></institution-wrap><country>United States</country></aff></contrib><contrib contrib-type="senior_editor"><name><surname>Cui</surname><given-names>Qiang</given-names></name><role>Senior Editor</role><aff><institution-wrap><institution-id institution-id-type="ror">https://ror.org/05qwgg493</institution-id><institution>Boston University</institution></institution-wrap><country>United States</country></aff></contrib></contrib-group><pub-date publication-format="electronic" date-type="publication"><day>30</day><month>10</month><year>2024</year></pub-date><volume>12</volume><elocation-id>RP88958</elocation-id><history><date date-type="sent-for-review" iso-8601-date="2023-05-09"><day>09</day><month>05</month><year>2023</year></date></history><pub-history><event><event-desc>This manuscript was published as a preprint.</event-desc><date date-type="preprint" iso-8601-date="2023-02-03"><day>03</day><month>02</month><year>2023</year></date><self-uri content-type="preprint" xlink:href="https://doi.org/10.1101/2023.02.02.526886"/></event><event><event-desc>This manuscript was published as a reviewed preprint.</event-desc><date date-type="reviewed-preprint" iso-8601-date="2023-07-18"><day>18</day><month>07</month><year>2023</year></date><self-uri content-type="reviewed-preprint" xlink:href="https://doi.org/10.7554/eLife.88958.1"/></event><event><event-desc>The reviewed preprint was revised.</event-desc><date date-type="reviewed-preprint" iso-8601-date="2024-09-17"><day>17</day><month>09</month><year>2024</year></date><self-uri content-type="reviewed-preprint" xlink:href="https://doi.org/10.7554/eLife.88958.2"/></event></pub-history><permissions><copyright-statement>© 2023, Qin and Zhou</copyright-statement><copyright-year>2023</copyright-year><copyright-holder>Qin and Zhou</copyright-holder><ali:free_to_read/><license xlink:href="http://creativecommons.org/licenses/by/4.0/"><ali:license_ref>http://creativecommons.org/licenses/by/4.0/</ali:license_ref><license-p>This article is distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="http://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution License</ext-link>, which permits unrestricted use and redistribution provided that the original author and source are credited.</license-p></license></permissions><self-uri content-type="pdf" xlink:href="elife-88958-v1.pdf"/><self-uri content-type="figures-pdf" xlink:href="elife-88958-figures-v1.pdf"/><abstract><p>How the sequences of intrinsically disordered proteins (IDPs) code for functions is still an enigma. Dynamics, in particular residue-specific dynamics, holds crucial clues. Enormous efforts have been spent to characterize residue-specific dynamics of IDPs, mainly through NMR spin relaxation experiments. Here, we present a sequence-based method, SeqDYN, for predicting residue-specific backbone dynamics of IDPs. SeqDYN employs a mathematical model with 21 parameters: one is a correlation length and 20 are the contributions of the amino acids to slow dynamics. Training on a set of 45 IDPs reveals aromatic, Arg, and long-branched aliphatic amino acids as the most active in slow dynamics whereas Gly and short polar amino acids as the least active. SeqDYN predictions not only provide an accurate and insightful characterization of sequence-dependent IDP dynamics but may also serve as indicators in a host of biophysical processes, including the propensities of IDP sequences to undergo phase separation.</p></abstract><kwd-group kwd-group-type="author-keywords"><kwd>NMR spectroscopy</kwd><kwd>phase separation</kwd><kwd>intrinsically disordered proteins</kwd><kwd>backbone dynamics</kwd><kwd>NMR spin relaxation</kwd></kwd-group><kwd-group kwd-group-type="research-organism"><title>Research organism</title><kwd>None</kwd></kwd-group><funding-group><award-group id="fund1"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100000002</institution-id><institution>National Institutes of Health</institution></institution-wrap></funding-source><award-id>GM118091</award-id><principal-award-recipient><name><surname>Zhou</surname><given-names>Huan-Xiang</given-names></name></principal-award-recipient></award-group><funding-statement>The funders had no role in study design, data collection and interpretation, or the decision to submit the work for publication.</funding-statement></funding-group><custom-meta-group><custom-meta specific-use="meta-only"><meta-name>Author impact statement</meta-name><meta-value>SeqDYN, a sequence-based method, accurately predicts the residue-specific transverse relaxation rates of intrinsically disordered proteins.</meta-value></custom-meta><custom-meta specific-use="meta-only"><meta-name>publishing-route</meta-name><meta-value>prc</meta-value></custom-meta></custom-meta-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>Intrinsically disordered proteins (IDPs) or regions (IDRs) do not have the luxury of a three-dimensional structure to help decipher the relationship between sequence and function. Instead, dynamics has emerged as a crucial link between sequence and function for IDPs (<xref ref-type="bibr" rid="bib19">Dey et al., 2022</xref>). Nuclear magnetic resonance (NMR) spin relaxation is a uniquely powerful technique for characterizing IDP dynamics, capable of yielding residue-specific information (<xref ref-type="bibr" rid="bib9">Camacho-Zarco et al., 2022</xref>). Backbone <sup>15</sup>N relaxation experiments typically yield three parameters per residue: transverse relaxation rate (<inline-formula><mml:math id="inf1"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula>), longitudinal relaxation rate (<inline-formula><mml:math id="inf2"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula>), and steady-state heteronuclear Overhauser enhancement (NOE). While all three parameters depend on ps-ns dynamics, <inline-formula><mml:math id="inf3"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> is the one most affected by slower dynamics (10 s of ns to 1 μs). An increase in either the timescale or the amplitude of slower dynamics results in higher <inline-formula><mml:math id="inf4"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula>. For IDPs, <inline-formula><mml:math id="inf5"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> is also the parameter that exhibits the strongest dependence on sequence (<xref ref-type="bibr" rid="bib19">Dey et al., 2022</xref>; <xref ref-type="bibr" rid="bib9">Camacho-Zarco et al., 2022</xref>).</p><p><inline-formula><mml:math id="inf6"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> was noted early on as an important indicator of residual structure in the unfolded state of the structured protein lysozyme (<xref ref-type="bibr" rid="bib31">Klein-Seetharaman et al., 2002</xref>). This property has since been measured for many IDPs to provide insight into various biophysical processes. Just as the residual structure in the unfolded state biases the folding pathway of lysozyme (<xref ref-type="bibr" rid="bib31">Klein-Seetharaman et al., 2002</xref>), a nascent α-helix in the free state of Sendai virus nucleoprotein C-terminal domain (Sev-NT), as indicated by highly elevated <inline-formula><mml:math id="inf7"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> (<xref ref-type="bibr" rid="bib1">Abyzov et al., 2016</xref>), biases the coupled binding and folding pathway in the presence of its target phosphoprotein (<xref ref-type="bibr" rid="bib60">Schneider et al., 2015</xref>). Local secondary structure preformation also facilitates the binding of yes-associated protein (YAP) with its target transcription factor (<xref ref-type="bibr" rid="bib21">Feichtinger et al., 2022</xref>). Likewise a correlation has been found between <inline-formula><mml:math id="inf8"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> in the free state and the membrane binding propensity of synaptobrevin-2: residues with elevated <inline-formula><mml:math id="inf9"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> have increased propensity for membrane binding (<xref ref-type="bibr" rid="bib32">Lakomek et al., 2019</xref>). <inline-formula><mml:math id="inf10"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> in the free state has also been used to uncover factors that promote liquid-liquid phase separation of IDPs. For example, a nascent α-helix (shown by elevated <inline-formula><mml:math id="inf11"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula>) is important for the phase separation of the TDP-43 low-complexity domain, as both the deletion of the helical region and a helix-breaking mutation (Ala to Pro) abrogates phase separation (<xref ref-type="bibr" rid="bib14">Conicella et al., 2016</xref>). Similarly, nascent α-helices in the free state of cytosolic abundant heat-soluble 8 (CAHS-8), upon raising concentration and lowering temperature stabilize to form the core of fibrous gels (<xref ref-type="bibr" rid="bib38">Malki et al., 2022</xref>). For the hnRNPA1 low-complexity domain (A1-LCD), aromatic residues giving rise to local peaks in <inline-formula><mml:math id="inf12"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> also mediate phase separation (<xref ref-type="bibr" rid="bib40">Martin et al., 2020</xref>).</p><p>Both NMR relaxation data and molecular dynamics (MD) simulations have revealed determinants of <inline-formula><mml:math id="inf13"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> for IDPs. It has been noted that the flexible Gly tends to lower <inline-formula><mml:math id="inf14"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula>, whereas secondary structure and contact formation tend to raise <inline-formula><mml:math id="inf15"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> (<xref ref-type="bibr" rid="bib15">Cook et al., 2019</xref>). This conclusion agrees well with recent MD simulations (<xref ref-type="bibr" rid="bib19">Dey et al., 2022</xref>; <xref ref-type="bibr" rid="bib25">Hicks et al., 2020</xref>; <xref ref-type="bibr" rid="bib80">Yu and Brüschweiler, 2022</xref>; <xref ref-type="bibr" rid="bib65">Smrt et al., 2023</xref>). These MD studies, using IDP-specific force fields, are able to predict <inline-formula><mml:math id="inf16"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> in quantitative agreement with NMR measurements, without ad hoc reweighting as done in earlier studies. According to MD, most contact clusters are formed by local sequences, within blocks of up to a dozen or so residues (<xref ref-type="bibr" rid="bib19">Dey et al., 2022</xref>; <xref ref-type="bibr" rid="bib25">Hicks et al., 2020</xref>; <xref ref-type="bibr" rid="bib65">Smrt et al., 2023</xref>). Tertiary contacts can also form but are relatively rare; as such their accurate capture requires extremely extensive sampling and still poses a challenge for MD simulations. Contrary to Gly, aromatic residues have been noted as mediators of contact clusters (<xref ref-type="bibr" rid="bib31">Klein-Seetharaman et al., 2002</xref>; <xref ref-type="bibr" rid="bib40">Martin et al., 2020</xref>).</p><p><xref ref-type="bibr" rid="bib61">Schwalbe et al., 1997</xref> introduced a mathematical model to describe the <inline-formula><mml:math id="inf17"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> profile along the sequence for lysozyme in the unfolded state. The <inline-formula><mml:math id="inf18"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> value of a given residue was expressed as the sum of contributions from this residue and its neighbors. This model yields a mostly flat profile across the sequence, except for a falloff at the termini, resulting in an overall bell shape. <xref ref-type="bibr" rid="bib31">Klein-Seetharaman et al., 2002</xref> then fit peaks above this flat profile as a sum of Gaussians. <xref ref-type="bibr" rid="bib11">Cho et al., 2007</xref> proposed bulkiness as a qualitative indicator of backbone dynamics. Recently <xref ref-type="bibr" rid="bib63">Sekiyama et al., 2022</xref> calculated <inline-formula><mml:math id="inf19"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> as the geometric mean of ‘indices of local dynamics’; the latter were parameterized by fitting to the measured <inline-formula><mml:math id="inf20"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> for a single IDP. All these models merely describe the <inline-formula><mml:math id="inf21"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> profile of a given IDP, and none of them is predictive.</p><p>Here, we present a method, SeqDYN, for predicting <inline-formula><mml:math id="inf22"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> of IDPs. Using a mathematical model introduced by <xref ref-type="bibr" rid="bib35">Li et al., 2020</xref> to predict propensities for binding nanoparticles and also adapted for predicting propensities for binding membranes (<xref ref-type="bibr" rid="bib52">Qin et al., 2022</xref>), we express the <inline-formula><mml:math id="inf23"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> value of a residue as the product of contributing factors from all residues. The contributing factor attenuates as the neighboring residue becomes more distant from the central residue. The model, after training on a set of 45 IDPs, has prediction accuracy that is competitive against that of the recent MD simulations using IDP-specific force fields (<xref ref-type="bibr" rid="bib19">Dey et al., 2022</xref>; <xref ref-type="bibr" rid="bib25">Hicks et al., 2020</xref>; <xref ref-type="bibr" rid="bib80">Yu and Brüschweiler, 2022</xref>; <xref ref-type="bibr" rid="bib65">Smrt et al., 2023</xref>). For lysozyme and other structured proteins, the SeqDYN prediction agrees remarkably well with <inline-formula><mml:math id="inf24"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> measured in their unfolded state.</p></sec><sec id="s2" sec-type="results"><title>Results</title><sec id="s2-1"><title>The data set of IDPs with <italic>R</italic><sub>2</sub> rates</title><p>We collected <italic>R</italic><sub>2</sub> data for a total of 54 nonhomologous IDPs or IDRs (<xref ref-type="table" rid="table1">Table 1</xref>; <xref ref-type="fig" rid="fig1">Figure 1</xref>). According to indicators from NMR properties, including low or negative NOEs, narrow dispersion in backbone amide proton chemical shifts, and small secondary chemical shifts (SCSs), most of the proteins are disordered with at most transient α-helices. A few are partially folded, including Sev-NT with a well-populated (~80%) long helix (residues 478–491; <xref ref-type="bibr" rid="bib29">Jensen et al., 2008</xref>), CREB-binding protein fourth intrinsically disordered linker (CBP-ID4) with &gt;50% propensities for two long helices (residues 2–25 and 101–128; <xref ref-type="bibr" rid="bib51">Piai et al., 2016</xref>), HOX transcription factor DFD (HOX-DFD) with a well-folded domain comprising three helices (<xref ref-type="bibr" rid="bib37">Maiti et al., 2019</xref>), and Hahellin (apo form) as a molten globule (<xref ref-type="bibr" rid="bib50">Patel et al., 2014</xref>). In <xref ref-type="fig" rid="fig2">Figure 2</xref>, we display representative conformations of five IDPs, ranging from fully disordered MAPK kinase 4 (MKK4; <xref ref-type="bibr" rid="bib18">Delaforge et al., 2018</xref>) and α-synuclein (<xref ref-type="bibr" rid="bib68">Sung and Eliezer, 2007</xref>) to Measles virus phosphoprotein N-terminal domain (Mev-P<sub>NTD</sub>; <xref ref-type="bibr" rid="bib43">Milles et al., 2018</xref>) with transient short helices to Sev-NT and CBP-ID4 with stable long helices. The sequences of all the IDPs are listed in Appendix 1.</p><fig id="fig1" position="float"><label>Figure 1.</label><caption><title>Clock-like tree plot showing lack of homology among the 45 IDPs.</title><p>The level of homology between two sequences is measured by the distance from their convergence point to the center of the clock. The highest level of apparent identity is between A1-LCD and TDP-43, at 25%, but these two proteins differ in both secondary structure formation and <inline-formula><mml:math id="inf25"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> characteristics. There is, however, a 20-residue overlap between the N-terminus of MBP-xα2 and the C-terminus of rmBG21.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-88958-fig1-v1.tif"/></fig><fig id="fig2" position="float"><label>Figure 2.</label><caption><title>Representative conformations of five IDPs.</title><p>(<bold>A–E</bold>) MKK4, α-synuclein, Mev-P<sub>NTD</sub>, Sev-NT, and CBP-ID4. Conformations were initially generated using TraDES (<ext-link ext-link-type="uri" xlink:href="http://trades.blueprint.org">http://trades.blueprint.org</ext-link>; <xref ref-type="bibr" rid="bib22">Feldman and Hogue, 2002</xref>), selected to have radius of gyration close to predicted by a scaling function <inline-formula><mml:math id="inf26"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mi>g</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mn>2.54</mml:mn><mml:msup><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mn>0.522</mml:mn></mml:mrow></mml:msup></mml:math></inline-formula> (Å) (<xref ref-type="bibr" rid="bib6">Bernadó and Blackledge, 2009</xref>). Conformations for residues predicted as helical by PsiPred plus filtering were replaced by an ideal helix. Finally residues are colored according to a scheme ranging from green for low predicted <inline-formula><mml:math id="inf27"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> to red for high predicted <inline-formula><mml:math id="inf28"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula>.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-88958-fig2-v1.tif"/></fig><table-wrap id="table1" position="float"><label>Table 1.</label><caption><title>Experimental conditions, mean and standard deviation of measured <inline-formula><mml:math id="inf29"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula>, and SeqDYN prediction RMSE.</title></caption><table frame="hsides" rules="groups"><thead><tr><th align="left" valign="bottom">Protein name</th><th align="left" valign="bottom"># of res</th><th align="left" valign="bottom">Temp(K)</th><th align="left" valign="bottom">B<sub>0</sub> (MHz)</th><th align="left" valign="bottom"><inline-formula><mml:math id="inf30"><mml:msub><mml:mrow><mml:mover accent="false"><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mo>¯</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula>(s<sup>–1</sup>)</th><th align="left" valign="bottom"><inline-formula><mml:math id="inf31"><mml:msub><mml:mrow><mml:mi>σ</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:mrow></mml:msub></mml:math></inline-formula> (s<sup>–1</sup>)</th><th align="left" valign="bottom">RMSE(s<sup>–1</sup>)</th><th align="left" valign="bottom">PMID; ref</th></tr></thead><tbody><tr><td align="left" valign="bottom" colspan="8">Training set (45 IDPs) <xref ref-type="table-fn" rid="table1fn1"><sup>*</sup></xref></td></tr><tr><td align="left" valign="bottom">A1-LCD</td><td align="left" valign="bottom">131</td><td align="left" valign="bottom">298</td><td align="left" valign="bottom">800</td><td align="left" valign="bottom">2.68</td><td align="left" valign="bottom">0.46</td><td align="left" valign="bottom">0.60</td><td align="left" valign="bottom"><ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/32029630/">32029630</ext-link>; <xref ref-type="bibr" rid="bib40">Martin et al., 2020</xref></td></tr><tr><td align="left" valign="bottom">Aβ40</td><td align="left" valign="bottom">40</td><td align="left" valign="bottom">278</td><td align="left" valign="bottom">600</td><td align="left" valign="bottom">3.40</td><td align="left" valign="bottom">0.92</td><td align="left" valign="bottom">0.38</td><td align="left" valign="bottom"><ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/31181936/">31181936</ext-link>; <xref ref-type="bibr" rid="bib55">Rezaei-Ghaleh et al., 2019</xref></td></tr><tr><td align="left" valign="bottom">Ash1</td><td align="left" valign="bottom">83</td><td align="left" valign="bottom">278</td><td align="left" valign="bottom">800</td><td align="left" valign="bottom">9.80</td><td align="left" valign="bottom">1.40</td><td align="left" valign="bottom">1.41</td><td align="left" valign="bottom"><ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/27807972/">27807972</ext-link>; <xref ref-type="bibr" rid="bib39">Martin et al., 2016</xref></td></tr><tr><td align="left" valign="bottom">Beclin1</td><td align="left" valign="bottom">165</td><td align="left" valign="bottom">288</td><td align="left" valign="bottom">800</td><td align="left" valign="bottom">5.37</td><td align="left" valign="bottom">1.03</td><td align="left" valign="bottom">1.14</td><td align="left" valign="bottom"><ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/27288992/">27288992</ext-link>; <xref ref-type="bibr" rid="bib79">Yao et al., 2016</xref></td></tr><tr><td align="left" valign="bottom">CAPRIN1</td><td align="left" valign="bottom">103</td><td align="left" valign="bottom">303</td><td align="left" valign="bottom">600</td><td align="left" valign="bottom">5.34</td><td align="left" valign="bottom">0.88</td><td align="left" valign="bottom">0.72</td><td align="left" valign="bottom"><ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/31898464/">31898464</ext-link>; <xref ref-type="bibr" rid="bib77">Wong et al., 2020</xref></td></tr><tr><td align="left" valign="bottom">CBP-ID4</td><td align="left" valign="bottom">207</td><td align="left" valign="bottom">283</td><td align="left" valign="bottom">700</td><td align="left" valign="bottom">5.45</td><td align="left" valign="bottom">2.55</td><td align="left" valign="bottom">2.01;1.90<xref ref-type="table-fn" rid="table1fn2"><sup>†</sup></xref></td><td align="left" valign="bottom"><ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/29790640/">29790640</ext-link>; <xref ref-type="bibr" rid="bib46">Murrali et al., 2018</xref></td></tr><tr><td align="left" valign="bottom">GbnD4-DHD</td><td align="left" valign="bottom">91</td><td align="left" valign="bottom">280</td><td align="left" valign="bottom">700</td><td align="left" valign="bottom">6.81</td><td align="left" valign="bottom">1.55</td><td align="left" valign="bottom">1.28</td><td align="left" valign="bottom"><ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/29309054/">29309054</ext-link>; <xref ref-type="bibr" rid="bib28">Jenner et al., 2018</xref></td></tr><tr><td align="left" valign="bottom">ERD14</td><td align="left" valign="bottom">185</td><td align="left" valign="bottom">288</td><td align="left" valign="bottom">600</td><td align="left" valign="bottom">3.96</td><td align="left" valign="bottom">0.87</td><td align="left" valign="bottom">0.54</td><td align="left" valign="bottom"><ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/21336827/">21336827</ext-link>; <xref ref-type="bibr" rid="bib69">Szalainé Ágoston et al., 2011</xref></td></tr><tr><td align="left" valign="bottom">ExsE</td><td align="left" valign="bottom">88</td><td align="left" valign="bottom">298</td><td align="left" valign="bottom">600</td><td align="left" valign="bottom">3.18</td><td align="left" valign="bottom">0.88</td><td align="left" valign="bottom">0.76</td><td align="left" valign="bottom"><ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/22138394/">22138394</ext-link>; <xref ref-type="bibr" rid="bib82">Zheng et al., 2012</xref></td></tr><tr><td align="left" valign="bottom">FCP1</td><td align="left" valign="bottom">85</td><td align="left" valign="bottom">298</td><td align="left" valign="bottom">500</td><td align="left" valign="bottom">2.94</td><td align="left" valign="bottom">0.54</td><td align="left" valign="bottom">0.43</td><td align="left" valign="bottom"><ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/26286791/">26286791</ext-link>; <xref ref-type="bibr" rid="bib34">Lawrence and Showalter, 2012</xref></td></tr><tr><td align="left" valign="bottom">FUS</td><td align="left" valign="bottom">163</td><td align="left" valign="bottom">298</td><td align="left" valign="bottom">850</td><td align="left" valign="bottom">3.48</td><td align="left" valign="bottom">0.51</td><td align="left" valign="bottom">0.54</td><td align="left" valign="bottom"><ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/26455390/">26455390</ext-link>; <xref ref-type="bibr" rid="bib8">Burke et al., 2015</xref></td></tr><tr><td align="left" valign="bottom">GAb1</td><td align="left" valign="bottom">82</td><td align="left" valign="bottom">298</td><td align="left" valign="bottom">500</td><td align="left" valign="bottom">3.99</td><td align="left" valign="bottom">0.88</td><td align="left" valign="bottom">0.89</td><td align="left" valign="bottom"><ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/34929201/">34929201</ext-link>; <xref ref-type="bibr" rid="bib23">Gruber et al., 2022</xref></td></tr><tr><td align="left" valign="bottom">hACTR</td><td align="left" valign="bottom">69</td><td align="left" valign="bottom">304</td><td align="left" valign="bottom">600</td><td align="left" valign="bottom">3.26</td><td align="left" valign="bottom">0.47</td><td align="left" valign="bottom">0.49</td><td align="left" valign="bottom"><ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/18177052/">18177052</ext-link>; <xref ref-type="bibr" rid="bib20">Ebert et al., 2008</xref></td></tr><tr><td align="left" valign="bottom">Hahellin</td><td align="left" valign="bottom">92</td><td align="left" valign="bottom">298</td><td align="left" valign="bottom">800</td><td align="left" valign="bottom">9.94</td><td align="left" valign="bottom">2.69</td><td align="left" valign="bottom">2.85</td><td align="left" valign="bottom"><ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/24671380/">24671380</ext-link>; <xref ref-type="bibr" rid="bib50">Patel et al., 2014</xref></td></tr><tr><td align="left" valign="bottom">hCSD1</td><td align="left" valign="bottom">141</td><td align="left" valign="bottom">298</td><td align="left" valign="bottom">500</td><td align="left" valign="bottom">3.56</td><td align="left" valign="bottom">0.93</td><td align="left" valign="bottom">0.99</td><td align="left" valign="bottom"><ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/18537264/">18537264</ext-link>; <xref ref-type="bibr" rid="bib30">Kiss et al., 2008</xref></td></tr><tr><td align="left" valign="bottom">HOX-DFD</td><td align="left" valign="bottom">90</td><td align="left" valign="bottom">298</td><td align="left" valign="bottom">600</td><td align="left" valign="bottom">6.98</td><td align="left" valign="bottom">3.15</td><td align="left" valign="bottom">1.99</td><td align="left" valign="bottom"><ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/30802457/">30802457</ext-link>; <xref ref-type="bibr" rid="bib37">Maiti et al., 2019</xref></td></tr><tr><td align="left" valign="bottom">hZIP4-ICL2</td><td align="left" valign="bottom">100</td><td align="left" valign="bottom">283</td><td align="left" valign="bottom">800</td><td align="left" valign="bottom">9.54</td><td align="left" valign="bottom">2.37</td><td align="left" valign="bottom">1.58</td><td align="left" valign="bottom"><ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/30793391/">30793391</ext-link>; <xref ref-type="bibr" rid="bib3">Bafaro et al., 2019</xref></td></tr><tr><td align="left" valign="bottom">Jaburetox</td><td align="left" valign="bottom">94</td><td align="left" valign="bottom">298</td><td align="left" valign="bottom">800</td><td align="left" valign="bottom">6.01</td><td align="left" valign="bottom">2.30</td><td align="left" valign="bottom">2.27</td><td align="left" valign="bottom"><ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/25605001/">25605001</ext-link>; <xref ref-type="bibr" rid="bib36">Lopes et al., 2015</xref></td></tr><tr><td align="left" valign="bottom">KRS-NT</td><td align="left" valign="bottom">72</td><td align="left" valign="bottom">303</td><td align="left" valign="bottom">600</td><td align="left" valign="bottom">3.26</td><td align="left" valign="bottom">0.93</td><td align="left" valign="bottom">0.83</td><td align="left" valign="bottom"><ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/24983501/">24983501</ext-link>; <xref ref-type="bibr" rid="bib12">Cho et al., 2014</xref></td></tr><tr><td align="left" valign="bottom">MBP-xα2</td><td align="left" valign="bottom">70</td><td align="left" valign="bottom">295</td><td align="left" valign="bottom">600</td><td align="left" valign="bottom">3.83</td><td align="left" valign="bottom">0.60</td><td align="left" valign="bottom">0.54</td><td align="left" valign="bottom"><ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/25343306/">25343306</ext-link>; <xref ref-type="bibr" rid="bib17">De Avila et al., 2014</xref></td></tr><tr><td align="left" valign="bottom">MKK4</td><td align="left" valign="bottom">86</td><td align="left" valign="bottom">278</td><td align="left" valign="bottom">850</td><td align="left" valign="bottom">4.49</td><td align="left" valign="bottom">1.42</td><td align="left" valign="bottom">0.63</td><td align="left" valign="bottom"><ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/29276882/">29276882</ext-link>; <xref ref-type="bibr" rid="bib18">Delaforge et al., 2018</xref></td></tr><tr><td align="left" valign="bottom">N-Cby</td><td align="left" valign="bottom">63</td><td align="left" valign="bottom">298</td><td align="left" valign="bottom"> </td><td align="left" valign="bottom">4.19</td><td align="left" valign="bottom">1.20</td><td align="left" valign="bottom">1.25</td><td align="left" valign="bottom"><ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/21182262/">21182262</ext-link>; <xref ref-type="bibr" rid="bib45">Mokhtarzada et al., 2011</xref></td></tr><tr><td align="left" valign="bottom">Niv-P<sub>NTD</sub></td><td align="left" valign="bottom">406</td><td align="left" valign="bottom">288</td><td align="left" valign="bottom">700</td><td align="left" valign="bottom">5.41</td><td align="left" valign="bottom">1.82</td><td align="left" valign="bottom">1.66</td><td align="left" valign="bottom"><ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/33177626/">33177626</ext-link>; <xref ref-type="bibr" rid="bib59">Schiavina et al., 2020</xref></td></tr><tr><td align="left" valign="bottom">NS5A-D2D3</td><td align="left" valign="bottom">268</td><td align="left" valign="bottom">278</td><td align="left" valign="bottom">800</td><td align="left" valign="bottom">8.62</td><td align="left" valign="bottom">3.85</td><td align="left" valign="bottom">2.14</td><td align="left" valign="bottom"><ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/26445449/">26445449</ext-link>; <xref ref-type="bibr" rid="bib66">Sólyom et al., 2015</xref></td></tr><tr><td align="left" valign="bottom">NUPR1</td><td align="left" valign="bottom">93</td><td align="left" valign="bottom">298</td><td align="left" valign="bottom">600</td><td align="left" valign="bottom">2.98</td><td align="left" valign="bottom">0.82</td><td align="left" valign="bottom">0.76</td><td align="left" valign="bottom"><ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/31325636/">31325636</ext-link>; <xref ref-type="bibr" rid="bib47">Neira et al., 2019</xref></td></tr><tr><td align="left" valign="bottom">OPN</td><td align="left" valign="bottom">220</td><td align="left" valign="bottom">310</td><td align="left" valign="bottom">800</td><td align="left" valign="bottom">2.59</td><td align="left" valign="bottom">0.82</td><td align="left" valign="bottom">0.54</td><td align="left" valign="bottom"><ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/31794728/">31794728</ext-link>; <xref ref-type="bibr" rid="bib41">Mateos et al., 2020</xref></td></tr><tr><td align="left" valign="bottom">p53TAD</td><td align="left" valign="bottom">73</td><td align="left" valign="bottom">298</td><td align="left" valign="bottom">850</td><td align="left" valign="bottom">2.72</td><td align="left" valign="bottom">0.66</td><td align="left" valign="bottom">0.33</td><td align="left" valign="bottom"><ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/30240067/">30240067</ext-link>; <xref ref-type="bibr" rid="bib78">Xie et al., 2018</xref></td></tr><tr><td align="left" valign="bottom">PDEγ</td><td align="left" valign="bottom">87</td><td align="left" valign="bottom">298</td><td align="left" valign="bottom"> </td><td align="left" valign="bottom">3.96</td><td align="left" valign="bottom">1.05</td><td align="left" valign="bottom">0.71</td><td align="left" valign="bottom"><ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/18230733/">18230733</ext-link>; <xref ref-type="bibr" rid="bib67">Song et al., 2008</xref></td></tr><tr><td align="left" valign="bottom">PKIα</td><td align="left" valign="bottom">75</td><td align="left" valign="bottom">300</td><td align="left" valign="bottom">900</td><td align="left" valign="bottom">3.41</td><td align="left" valign="bottom">0.87</td><td align="left" valign="bottom">0.52</td><td align="left" valign="bottom"><ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/32338601/">32338601</ext-link>; <xref ref-type="bibr" rid="bib48">Olivieri et al., 2020</xref></td></tr><tr><td align="left" valign="bottom">Mev-P<sub>NTD</sub></td><td align="left" valign="bottom">304</td><td align="left" valign="bottom">298</td><td align="left" valign="bottom">950</td><td align="left" valign="bottom">2.92</td><td align="left" valign="bottom">0.59</td><td align="left" valign="bottom">0.48</td><td align="left" valign="bottom"><ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/30140745/">30140745</ext-link>; <xref ref-type="bibr" rid="bib43">Milles et al., 2018</xref></td></tr><tr><td align="left" valign="bottom">ProTα</td><td align="left" valign="bottom">113</td><td align="left" valign="bottom">283</td><td align="left" valign="bottom">800</td><td align="left" valign="bottom">3.40</td><td align="left" valign="bottom">0.56</td><td align="left" valign="bottom">0.43</td><td align="left" valign="bottom"><ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/29466338/">29466338</ext-link>; <xref ref-type="bibr" rid="bib7">Borgia et al., 2018</xref></td></tr><tr><td align="left" valign="bottom">Pup</td><td align="left" valign="bottom">64</td><td align="left" valign="bottom">298</td><td align="left" valign="bottom">850</td><td align="left" valign="bottom">2.66</td><td align="left" valign="bottom">0.51</td><td align="left" valign="bottom">0.43</td><td align="left" valign="bottom"><ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/30240067/">30240067</ext-link>; <xref ref-type="bibr" rid="bib78">Xie et al., 2018</xref></td></tr><tr><td align="left" valign="bottom">rmBG21</td><td align="left" valign="bottom">199</td><td align="left" valign="bottom">300</td><td align="left" valign="bottom">600</td><td align="left" valign="bottom">4.06</td><td align="left" valign="bottom">0.90</td><td align="left" valign="bottom">0.63</td><td align="left" valign="bottom"><ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/17676872/">17676872</ext-link>; <xref ref-type="bibr" rid="bib2">Ahmed et al., 2007</xref></td></tr><tr><td align="left" valign="bottom">RPB1</td><td align="left" valign="bottom">201</td><td align="left" valign="bottom">277</td><td align="left" valign="bottom">850</td><td align="left" valign="bottom">6.48</td><td align="left" valign="bottom">1.74</td><td align="left" valign="bottom">1.33</td><td align="left" valign="bottom"><ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/28945358/">28945358</ext-link>; <xref ref-type="bibr" rid="bib27">Janke et al., 2018</xref></td></tr><tr><td align="left" valign="bottom">securin</td><td align="left" valign="bottom">202</td><td align="left" valign="bottom">283</td><td align="left" valign="bottom">500</td><td align="left" valign="bottom">5.49</td><td align="left" valign="bottom">1.13</td><td align="left" valign="bottom">1.08</td><td align="left" valign="bottom"><ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/19053469/">19053469</ext-link>; <xref ref-type="bibr" rid="bib16">Csizmok et al., 2008</xref></td></tr><tr><td align="left" valign="bottom">Sev-NT</td><td align="left" valign="bottom">124</td><td align="left" valign="bottom">298</td><td align="left" valign="bottom">600</td><td align="left" valign="bottom">3.20</td><td align="left" valign="bottom">1.42</td><td align="left" valign="bottom">0.76;0.38<xref ref-type="table-fn" rid="table1fn2"><sup>†</sup></xref></td><td align="left" valign="bottom"><ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/27112095/">27112095</ext-link>; <xref ref-type="bibr" rid="bib1">Abyzov et al., 2016</xref></td></tr><tr><td align="left" valign="bottom">Sic1</td><td align="left" valign="bottom">92</td><td align="left" valign="bottom">278</td><td align="left" valign="bottom">500</td><td align="left" valign="bottom">3.34</td><td align="left" valign="bottom">0.59</td><td align="left" valign="bottom">0.48</td><td align="left" valign="bottom"><ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/20399186/">20399186</ext-link>; <xref ref-type="bibr" rid="bib44">Mittag et al., 2010</xref></td></tr><tr><td align="left" valign="bottom">SKIPN</td><td align="left" valign="bottom">71</td><td align="left" valign="bottom">298</td><td align="left" valign="bottom"> </td><td align="left" valign="bottom">5.64</td><td align="left" valign="bottom">1.05</td><td align="left" valign="bottom">1.46</td><td align="left" valign="bottom"><ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/20007319/">20007319</ext-link>; <xref ref-type="bibr" rid="bib74">Wang et al., 2010</xref></td></tr><tr><td align="left" valign="bottom">SLBP-NT</td><td align="left" valign="bottom">113</td><td align="left" valign="bottom">298</td><td align="left" valign="bottom">600</td><td align="left" valign="bottom">3.96</td><td align="left" valign="bottom">1.40</td><td align="left" valign="bottom">1.61</td><td align="left" valign="bottom"><ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/15260482/">15260482</ext-link>; <xref ref-type="bibr" rid="bib72">Thapar et al., 2004</xref></td></tr><tr><td align="left" valign="bottom">α-synuclein</td><td align="left" valign="bottom">140</td><td align="left" valign="bottom">298</td><td align="left" valign="bottom">600</td><td align="left" valign="bottom">2.96</td><td align="left" valign="bottom">0.53</td><td align="left" valign="bottom">0.44</td><td align="left" valign="bottom"><ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/30184304/">30184304</ext-link>; <xref ref-type="bibr" rid="bib54">Rezaei-Ghaleh et al., 2018</xref></td></tr><tr><td align="left" valign="bottom">SOCS5-JIR</td><td align="left" valign="bottom">70</td><td align="left" valign="bottom">303</td><td align="left" valign="bottom">800</td><td align="left" valign="bottom">4.32</td><td align="left" valign="bottom">2.36</td><td align="left" valign="bottom">1.91</td><td align="left" valign="bottom"><ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/26173083/">26173083</ext-link>; <xref ref-type="bibr" rid="bib10">Chandrashekaran et al., 2015</xref></td></tr><tr><td align="left" valign="bottom">tau K18</td><td align="left" valign="bottom">129</td><td align="left" valign="bottom">283</td><td align="left" valign="bottom">700</td><td align="left" valign="bottom">4.12</td><td align="left" valign="bottom">0.95</td><td align="left" valign="bottom">0.83</td><td align="left" valign="bottom"><ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/23740819/">23740819</ext-link>; <xref ref-type="bibr" rid="bib5">Barré and Eliezer, 2013</xref></td></tr><tr><td align="left" valign="bottom">TC1</td><td align="left" valign="bottom">106</td><td align="left" valign="bottom">298</td><td align="left" valign="bottom">600</td><td align="left" valign="bottom">4.65</td><td align="left" valign="bottom">1.61</td><td align="left" valign="bottom">1.24</td><td align="left" valign="bottom"><ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/23189168/">23189168</ext-link>; <xref ref-type="bibr" rid="bib13">Cino et al., 2012</xref></td></tr><tr><td align="left" valign="bottom">TDP-43</td><td align="left" valign="bottom">151</td><td align="left" valign="bottom">283</td><td align="left" valign="bottom">500</td><td align="left" valign="bottom">4.07</td><td align="left" valign="bottom">1.51</td><td align="left" valign="bottom">0.96</td><td align="left" valign="bottom"><ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/27545621/">27545621</ext-link>; <xref ref-type="bibr" rid="bib14">Conicella et al., 2016</xref></td></tr><tr><td align="left" valign="bottom">γ-tubulin-CT</td><td align="left" valign="bottom">39</td><td align="left" valign="bottom">288</td><td align="left" valign="bottom">500</td><td align="left" valign="bottom">2.23</td><td align="left" valign="bottom">0.35</td><td align="left" valign="bottom">0.27</td><td align="left" valign="bottom"><ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/29127738/">29127738</ext-link>; <xref ref-type="bibr" rid="bib24">Harris et al., 2018</xref></td></tr><tr><td align="left" valign="bottom" colspan="8">Test set (9 IDPs)</td></tr><tr><td align="left" valign="bottom">AMOTL1</td><td align="left" valign="bottom">207</td><td align="left" valign="bottom">283</td><td align="left" valign="bottom">800</td><td align="left" valign="bottom">8.45</td><td align="left" valign="bottom">2.55</td><td align="left" valign="bottom">2.04</td><td align="left" valign="bottom"><ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/35481651/">35481651</ext-link>; <xref ref-type="bibr" rid="bib73">Vogel et al., 2022</xref></td></tr><tr><td align="left" valign="bottom">CAHS-8</td><td align="left" valign="bottom">233</td><td align="left" valign="bottom">303</td><td align="left" valign="bottom">850</td><td align="left" valign="bottom">4.43</td><td align="left" valign="bottom">3.25</td><td align="left" valign="bottom">2.36;1.92<xref ref-type="table-fn" rid="table1fn2"><sup>†</sup></xref></td><td align="left" valign="bottom"><ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/34750927/">34750927</ext-link>; <xref ref-type="bibr" rid="bib38">Malki et al., 2022</xref></td></tr><tr><td align="left" valign="bottom">ChiZ</td><td align="left" valign="bottom">64</td><td align="left" valign="bottom">298</td><td align="left" valign="bottom">800</td><td align="left" valign="bottom">4.33</td><td align="left" valign="bottom">0.89</td><td align="left" valign="bottom">0.74</td><td align="left" valign="bottom"><ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/32585849/">32585849</ext-link>; <xref ref-type="bibr" rid="bib25">Hicks et al., 2020</xref></td></tr><tr><td align="left" valign="bottom">α-endosulfine</td><td align="left" valign="bottom">121</td><td align="left" valign="bottom">298</td><td align="left" valign="bottom">800</td><td align="left" valign="bottom">3.21</td><td align="left" valign="bottom">0.81</td><td align="left" valign="bottom">0.48</td><td align="left" valign="bottom"><ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/34346186/">34346186</ext-link>; <xref ref-type="bibr" rid="bib71">Thapa et al., 2022</xref></td></tr><tr><td align="left" valign="bottom">FtsQ</td><td align="left" valign="bottom">99</td><td align="left" valign="bottom">305</td><td align="left" valign="bottom">800</td><td align="left" valign="bottom">6.44</td><td align="left" valign="bottom">3.78</td><td align="left" valign="bottom">2.32;1.71<xref ref-type="table-fn" rid="table1fn2"><sup>†</sup></xref></td><td align="left" valign="bottom"><ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/36959324/">36959324</ext-link>; <xref ref-type="bibr" rid="bib65">Smrt et al., 2023</xref></td></tr><tr><td align="left" valign="bottom">Pdx1</td><td align="left" valign="bottom">83</td><td align="left" valign="bottom">298</td><td align="left" valign="bottom">500</td><td align="left" valign="bottom">2.98</td><td align="left" valign="bottom">0.70</td><td align="left" valign="bottom">0.76</td><td align="left" valign="bottom"><ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/30525611/">30525611</ext-link>; <xref ref-type="bibr" rid="bib15">Cook et al., 2019</xref></td></tr><tr><td align="left" valign="bottom">synaptobrevin-2</td><td align="left" valign="top">96</td><td align="left" valign="bottom">278</td><td align="left" valign="bottom">600</td><td align="left" valign="bottom">5.54</td><td align="left" valign="bottom">1.80</td><td align="left" valign="bottom">0.72</td><td align="left" valign="bottom"><ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/30975750/">30975750</ext-link>; <xref ref-type="bibr" rid="bib32">Lakomek et al., 2019</xref></td></tr><tr><td align="left" valign="bottom">TIA-1</td><td align="left" valign="bottom">91</td><td align="left" valign="bottom">310</td><td align="left" valign="bottom">800</td><td align="left" valign="bottom">4.01</td><td align="left" valign="bottom">0.89</td><td align="left" valign="bottom">0.55</td><td align="left" valign="bottom"><ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/36112647/">36112647</ext-link>; <xref ref-type="bibr" rid="bib63">Sekiyama et al., 2022</xref></td></tr><tr><td align="left" valign="bottom">YAP</td><td align="left" valign="bottom">122</td><td align="left" valign="bottom">298</td><td align="left" valign="bottom">800</td><td align="left" valign="bottom">3.19</td><td align="left" valign="bottom">1.44</td><td align="left" valign="bottom">1.23</td><td align="left" valign="bottom"><ext-link ext-link-type="uri" xlink:href="https://pubmed.ncbi.nlm.nih.gov/35378854/">35378854</ext-link>; <xref ref-type="bibr" rid="bib21">Feichtinger et al., 2022</xref></td></tr></tbody></table><table-wrap-foot><fn id="table1fn1"><label>*</label><p>For training set, RMSE is calculated for prediction based on leave-one-out training (using 44 IDPs).</p></fn><fn id="table1fn2"><label>†</label><p>First number is for SeqDYN prediction; second number is after applying a helix boost.</p></fn></table-wrap-foot></table-wrap><p>We used 45 of the 54 IDPs to train and validate SeqDYN and reserved the remaining 9 for testing. The sequence lengths of the training set range from 39 to 406 residues, with an average of 125.3 residues. Altogether <inline-formula><mml:math id="inf32"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> data are available for 3966 residues. A large majority (35 out of 45) of the 45 IDPs have mean <inline-formula><mml:math id="inf33"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> values (<inline-formula><mml:math id="inf34"><mml:msub><mml:mrow><mml:mover accent="false"><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mo>¯</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula>, calculated among all the residues in a protein) between 2.5 and 5.5 s<sup>–1</sup> (<xref ref-type="table" rid="table1">Table 1</xref> and <xref ref-type="fig" rid="fig3">Figure 3A</xref>). This <inline-formula><mml:math id="inf35"><mml:msub><mml:mrow><mml:mover accent="false"><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mo>¯</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> range is much lower than that of structured proteins with similar sequence lengths. The low <inline-formula><mml:math id="inf36"><mml:msub><mml:mrow><mml:mover accent="false"><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mo>¯</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> values and lack of dependence on sequence length (<xref ref-type="fig" rid="fig3s1">Figure 3—figure supplement 1A</xref>) suggest that <inline-formula><mml:math id="inf37"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> of the IDPs is mostly dictated by local sequence instead of tertiary interaction.</p><fig-group><fig id="fig3" position="float"><label>Figure 3.</label><caption><title>Properties of the 45 IDPs in the training set.</title><p>(<bold>A</bold>) Histograms of means and standard deviations, calculated for individual proteins. Curves are drawn to guide the eye. Inset: correlation between <inline-formula><mml:math id="inf38"><mml:msub><mml:mrow><mml:mover accent="false"><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mo>¯</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> and <inline-formula><mml:math id="inf39"><mml:msub><mml:mrow><mml:mi>σ</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:mrow></mml:msub></mml:math></inline-formula>. (<bold>B</bold>) Experimental mean scaled <inline-formula><mml:math id="inf40"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> (<inline-formula><mml:math id="inf41"><mml:mi>m</mml:mi><mml:mi>s</mml:mi><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula>) and SeqDYN <inline-formula><mml:math id="inf42"><mml:mi>q</mml:mi></mml:math></inline-formula> parameters, for the 20 types of amino acids. Note that Pro residues have low <inline-formula><mml:math id="inf43"><mml:mi>m</mml:mi><mml:mi>s</mml:mi><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> for the lack of backbone amide proton. Amino acids are in descending order of <inline-formula><mml:math id="inf44"><mml:mi>q</mml:mi></mml:math></inline-formula>.</p><p><supplementary-material id="fig3sdata1"><label>Figure 3—source data 1.</label><caption><title>Source data for <xref ref-type="fig" rid="fig3">Figure 3</xref>.</title></caption><media mimetype="application" mime-subtype="xlsx" xlink:href="elife-88958-fig3-data1-v1.xlsx"/></supplementary-material></p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-88958-fig3-v1.tif"/></fig><fig id="fig3s1" position="float" specific-use="child-fig"><label>Figure 3—figure supplement 1.</label><caption><title>Possible effects of sequence length, temperature, and magnetic field on <inline-formula><mml:math id="inf45"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula>.</title><p>(<bold>A</bold>) Lack of dependence of <inline-formula><mml:math id="inf46"><mml:msub><mml:mrow><mml:mover accent="false"><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mo>¯</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> on sequence length. (<bold>B</bold>) Counts of IDPs with <inline-formula><mml:math id="inf47"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> measured at various temperatures. (<bold>C</bold>) Matching of <inline-formula><mml:math id="inf48"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> profiles of Sev-NT measured at two temperatures after uniform scaling. (<bold>D</bold>) Matching of <inline-formula><mml:math id="inf49"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> profiles of A1-LCD measured at two temperatures after uniform scaling. (<bold>E</bold>) Counts of IDPs with <inline-formula><mml:math id="inf50"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> measured at various magnetic fields.</p><p><supplementary-material id="fig3s1sdata1"><label>Figure 3—figure supplement 1—source data 1.</label><caption><title>Source data for <xref ref-type="fig" rid="fig3s1">Figure 3—figure supplement 1</xref>.</title></caption><media mimetype="application" mime-subtype="xlsx" xlink:href="elife-88958-fig3-figsupp1-data1-v1.xlsx"/></supplementary-material></p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-88958-fig3-figsupp1-v1.tif"/></fig></fig-group><p>The most often used temperature for acquiring the <inline-formula><mml:math id="inf51"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> data was 298 K, but low temperatures (277–280 K) were used in a few cases (<xref ref-type="table" rid="table1">Table 1</xref> and <xref ref-type="fig" rid="fig3s1">Figure 3—figure supplement 1B</xref>). Of the seven IDPs with <inline-formula><mml:math id="inf52"><mml:msub><mml:mrow><mml:mover accent="false"><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mo>¯</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> &gt; 6.4 s<sup>–1</sup>, four can be attributed to low temperatures (<xref ref-type="bibr" rid="bib66">Sólyom et al., 2015</xref>; <xref ref-type="bibr" rid="bib39">Martin et al., 2016</xref>; <xref ref-type="bibr" rid="bib27">Janke et al., 2018</xref>; <xref ref-type="bibr" rid="bib28">Jenner et al., 2018</xref>), one is due to a relatively low temperature (283 K) as well as the presence of glycerol (20% v/v; <xref ref-type="bibr" rid="bib3">Bafaro et al., 2019</xref>), and two can be explained by tertiary structure formation [a folded domain (<xref ref-type="bibr" rid="bib37">Maiti et al., 2019</xref>) or molten globule (<xref ref-type="bibr" rid="bib50">Patel et al., 2014</xref>)]. A simple reason for higher <inline-formula><mml:math id="inf53"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> values at lower temperatures is the higher water viscosity, resulting in a slowdown in molecular tumbling; a similar effect is achieved by adding glycerol. In some cases, <inline-formula><mml:math id="inf54"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> was measured at both low and room temperatures (<xref ref-type="bibr" rid="bib1">Abyzov et al., 2016</xref>; <xref ref-type="bibr" rid="bib40">Martin et al., 2020</xref>). To a good approximation, the effect of lowering temperature is a uniform scaling of <inline-formula><mml:math id="inf55"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> across the IDP sequence. For Sev-NT, downscaling of the <inline-formula><mml:math id="inf56"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> values at 278 K by a factor of 2.0 brings them into close agreement with those at 298 K (<xref ref-type="fig" rid="fig3s1">Figure 3—figure supplement 1C</xref>), with a root-mean-square-deviation (RMSD) of 0.5 s<sup>–1</sup> among all the residues. Likewise, for A1-LCD, downscaling by a factor of 2.4 brings the <inline-formula><mml:math id="inf57"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> values at 288 K into good match with those at 298 K (<xref ref-type="fig" rid="fig3s1">Figure 3—figure supplement 1D</xref>), with an RMSD of 0.4 s<sup>–1</sup>. Because SeqDYN is concerned with the sequence dependence of <inline-formula><mml:math id="inf58"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula>, a uniform scaling has no effect on model parameter or prediction; therefore mixing the data from different temperatures is justified. The same can be said about the different magnetic fields in acquiring the <inline-formula><mml:math id="inf59"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> data (<xref ref-type="table" rid="table1">Table 1</xref> and <xref ref-type="fig" rid="fig3s1">Figure 3—figure supplement 1E</xref>). Increasing the magnetic field raises <inline-formula><mml:math id="inf60"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> values, and the effect is also approximated well by a uniform scaling (<xref ref-type="bibr" rid="bib1">Abyzov et al., 2016</xref>; <xref ref-type="bibr" rid="bib14">Conicella et al., 2016</xref>; <xref ref-type="bibr" rid="bib27">Janke et al., 2018</xref>).</p><p>One measure on the level of sequence dependence of <inline-formula><mml:math id="inf61"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> is the standard deviation, <inline-formula><mml:math id="inf62"><mml:msub><mml:mrow><mml:mi>σ</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:mrow></mml:msub></mml:math></inline-formula>, calculated among the residues of an IDP. Among the training set, the <inline-formula><mml:math id="inf63"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> values of 30 IDPs have moderate sequence variations, with <inline-formula><mml:math id="inf64"><mml:msub><mml:mrow><mml:mi>σ</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:mrow></mml:msub></mml:math></inline-formula> ranging from 0.5 to 1.5 s<sup>–1</sup> (<xref ref-type="table" rid="table1">Table 1</xref>); the histogram of <inline-formula><mml:math id="inf65"><mml:msub><mml:mrow><mml:mi>σ</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:mrow></mml:msub></mml:math></inline-formula> calculated for the entire training set peaks around 0.75 s<sup>–1</sup> (<xref ref-type="fig" rid="fig3">Figure 3A</xref>). There is a moderate correlation between <inline-formula><mml:math id="inf66"><mml:msub><mml:mrow><mml:mi>σ</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:mrow></mml:msub></mml:math></inline-formula> and <inline-formula><mml:math id="inf67"><mml:msub><mml:mrow><mml:mover accent="false"><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mo>¯</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> (<xref ref-type="fig" rid="fig3">Figure 3A</xref>, inset), reflecting in part the fact that <inline-formula><mml:math id="inf68"><mml:msub><mml:mrow><mml:mi>σ</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:mrow></mml:msub></mml:math></inline-formula> can be raised simply by a uniform upscaling, for example as a result of lowering temperature. Still, only two of the five IDPs with high <inline-formula><mml:math id="inf69"><mml:msub><mml:mrow><mml:mover accent="false"><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mo>¯</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> attributable to lower temperature or presence of glycerol are among the seven IDPs with high sequence variations (<inline-formula><mml:math id="inf70"><mml:msub><mml:mrow><mml:mi>σ</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:mrow></mml:msub></mml:math></inline-formula>&gt;2 s<sup>–1</sup>). Therefore, the sequence variation of <inline-formula><mml:math id="inf71"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> as captured by <inline-formula><mml:math id="inf72"><mml:msub><mml:mrow><mml:mi>σ</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:mrow></mml:msub></mml:math></inline-formula> manifests mostly the intrinsic effect of the IDP sequence, not the influence of external factors such as temperature or magnetic field strength. The mean <inline-formula><mml:math id="inf73"><mml:msub><mml:mrow><mml:mi>σ</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:mrow></mml:msub></mml:math></inline-formula> value among the training set is 1.24 s<sup>–1</sup>.</p><p>One way to eliminate the influence of external factors is to scale the <inline-formula><mml:math id="inf74"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> values of each IDP by its <inline-formula><mml:math id="inf75"><mml:msub><mml:mrow><mml:mover accent="false"><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mo>¯</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula>; we refer to the results as scaled <inline-formula><mml:math id="inf76"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula>, or <inline-formula><mml:math id="inf77"><mml:mi>s</mml:mi><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula>. We then pooled the <inline-formula><mml:math id="inf78"><mml:mi>s</mml:mi><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> values for all residues in the training set, and separated them according to amino-acid types. The amino acid type-specific mean <inline-formula><mml:math id="inf79"><mml:mi>s</mml:mi><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> values, or <inline-formula><mml:math id="inf80"><mml:mi>m</mml:mi><mml:mi>s</mml:mi><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula>, are displayed in <xref ref-type="fig" rid="fig3">Figure 3B</xref>. The seven amino acids with the highest <inline-formula><mml:math id="inf81"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi mathvariant="normal">m</mml:mi><mml:mi mathvariant="normal">s</mml:mi><mml:mrow><mml:msub><mml:mi mathvariant="italic">R</mml:mi><mml:mn mathvariant="italic">2</mml:mn></mml:msub></mml:mrow></mml:mrow></mml:mstyle></mml:math></inline-formula>  in descending order are Trp, Arg, Tyr, Phe, Ile, His, and Leu. The presence of all the four aromatic amino acids in this “high-end” group immediately suggests π-π stacking as important for raising <inline-formula><mml:math id="inf82"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi mathvariant="normal">m</mml:mi><mml:mi mathvariant="normal">s</mml:mi><mml:mrow><mml:msub><mml:mi mathvariant="italic">R</mml:mi><mml:mn mathvariant="italic">2</mml:mn></mml:msub></mml:mrow></mml:mrow></mml:mstyle></mml:math></inline-formula>; the presence of Arg further implicates cation-π interactions. In the other extreme, the seven amino acids with the lowest <inline-formula><mml:math id="inf83"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi mathvariant="normal">m</mml:mi><mml:mi mathvariant="normal">s</mml:mi><mml:mrow><mml:msub><mml:mi mathvariant="italic">R</mml:mi><mml:mn mathvariant="italic">2</mml:mn></mml:msub></mml:mrow></mml:mrow></mml:mstyle></mml:math></inline-formula> in ascending order are Gly, Cys, Val, Asp, Ser, Thr, and Asn. Gly is well-known as a flexible residue; it is also interesting that all the four amino acids with short polar sidechains are found in this “low-end” group. Pro has an excessively low <inline-formula><mml:math id="inf84"><mml:mi>m</mml:mi><mml:mi>s</mml:mi><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> [with data from only two IDPs (<xref ref-type="bibr" rid="bib46">Murrali et al., 2018</xref>; <xref ref-type="bibr" rid="bib77">Wong et al., 2020</xref>)], but that is due to the absence of an amide proton.</p></sec><sec id="s2-2"><title>The SeqDYN model and parameters</title><p>The null model is to assume a uniform <inline-formula><mml:math id="inf85"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> for all the residues in an IDP. The root-mean-square-error (RMSE) of the null model is equal to the standard deviation, <inline-formula><mml:math id="inf86"><mml:msub><mml:mrow><mml:mi>σ</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:mrow></mml:msub></mml:math></inline-formula>, of the measured <inline-formula><mml:math id="inf87"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> values. The mean RMSE, <inline-formula><mml:math id="inf88"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mrow><mml:mrow><mml:mover><mml:mrow><mml:mi>R</mml:mi><mml:mi>M</mml:mi><mml:mi>S</mml:mi><mml:mi>E</mml:mi></mml:mrow><mml:mo stretchy="false">¯</mml:mo></mml:mover></mml:mrow></mml:mrow></mml:mrow></mml:mstyle></mml:math></inline-formula>, of the null model, equal to 1.24 s<sup>–1</sup> for the training set, serves as the upper bound for evaluating the errors of <inline-formula><mml:math id="inf89"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> predictors. The next improvement is a one-residue predictor, where first each residue (with index <italic>n</italic>) assumes its amino acid-specific mean <inline-formula><mml:math id="inf90"><mml:mi>s</mml:mi><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> (<inline-formula><mml:math id="inf91"><mml:mi>m</mml:mi><mml:mi>s</mml:mi><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula>) and then a uniform scaling factor <inline-formula><mml:math id="inf92"><mml:mi>Υ</mml:mi></mml:math></inline-formula> is applied:<disp-formula id="equ1"><label>(1)</label><mml:math id="m1"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mfenced separators="|"><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:mfenced><mml:mo>=</mml:mo><mml:mi>Υ</mml:mi><mml:mo>∙</mml:mo><mml:msub><mml:mrow><mml:mi>m</mml:mi><mml:mi>s</mml:mi><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mfenced separators="|"><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:mfenced></mml:math></disp-formula></p><p>This one-residue model does only minutely better than the null model, with a <inline-formula><mml:math id="inf93"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mrow><mml:mrow><mml:mover><mml:mrow><mml:mi>R</mml:mi><mml:mi>M</mml:mi><mml:mi>S</mml:mi><mml:mi>E</mml:mi></mml:mrow><mml:mo stretchy="false">¯</mml:mo></mml:mover></mml:mrow></mml:mrow></mml:mrow></mml:mstyle></mml:math></inline-formula> of 1.22 s<sup>–1</sup>.</p><p>In SeqDYN, we account for the influence of neighboring residues. Specifically, each residue <italic>i</italic> contributes a factor <inline-formula><mml:math id="inf94"><mml:mi>f</mml:mi><mml:mfenced separators="|"><mml:mrow><mml:mi>i</mml:mi><mml:mo>;</mml:mo><mml:mi>n</mml:mi></mml:mrow></mml:mfenced></mml:math></inline-formula> to the <inline-formula><mml:math id="inf95"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> value of residue <italic>n</italic>. Therefore,<disp-formula id="equ2"><label>(2a)</label><mml:math id="m2"><mml:mrow><mml:msub><mml:mi>R</mml:mi><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mrow><mml:mo>(</mml:mo><mml:mi>n</mml:mi><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mi mathvariant="normal">Υ</mml:mi><mml:munderover><mml:mo>∏</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>N</mml:mi></mml:mrow></mml:munderover><mml:mi>f</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>;</mml:mo><mml:mi>n</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:math></disp-formula></p><p>where <italic>N</italic> is the total number of residues in the IDP. The contributing factor depends on the sequence distance <inline-formula><mml:math id="inf96"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>s</mml:mi><mml:mo>=</mml:mo><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:mi>i</mml:mi><mml:mo>−</mml:mo><mml:mi>n</mml:mi><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow></mml:mrow></mml:mstyle></mml:math></inline-formula> and the amino-acid type of residue <inline-formula><mml:math id="inf97"><mml:mi>i</mml:mi></mml:math></inline-formula>:<disp-formula id="equ3"><label>(2b)</label><mml:math id="m3"><mml:mrow><mml:mi>f</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>;</mml:mo><mml:mi>n</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mn>1</mml:mn><mml:mo>+</mml:mo><mml:mfrac><mml:mrow><mml:mi>q</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mi>i</mml:mi><mml:mo>)</mml:mo></mml:mrow><mml:mo>−</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mn>1</mml:mn><mml:mo>+</mml:mo><mml:mi>b</mml:mi><mml:msup><mml:mi>s</mml:mi><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup></mml:mrow></mml:mfrac></mml:mrow></mml:math></disp-formula></p><p>There are 21 global parameters. The first 20 are the <inline-formula><mml:math id="inf98"><mml:mi>q</mml:mi></mml:math></inline-formula> values, one for each of the 20 types of amino acids; the last parameter is <inline-formula><mml:math id="inf99"><mml:mi>b</mml:mi></mml:math></inline-formula>, appearing in the Lorentzian form of the sequence-distance dependence. We define the correlation length, <inline-formula><mml:math id="inf100"><mml:msub><mml:mrow><mml:mi>L</mml:mi></mml:mrow><mml:mrow><mml:mi>c</mml:mi><mml:mi>o</mml:mi><mml:mi>r</mml:mi><mml:mi>r</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>, as the sequence distance at which the contributing factor is midway between the values at <inline-formula><mml:math id="inf101"><mml:mi>s</mml:mi><mml:mo>=</mml:mo><mml:mn>0</mml:mn></mml:math></inline-formula> and <inline-formula><mml:math id="inf102"><mml:mi>∞</mml:mi></mml:math></inline-formula>. It is easy to verify that <inline-formula><mml:math id="inf103"><mml:msub><mml:mrow><mml:mi>L</mml:mi></mml:mrow><mml:mrow><mml:mi>c</mml:mi><mml:mi>o</mml:mi><mml:mi>r</mml:mi><mml:mi>r</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msup><mml:mrow><mml:mi>b</mml:mi></mml:mrow><mml:mrow><mml:mrow><mml:mrow><mml:mo>-</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mo>/</mml:mo><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:mrow></mml:mrow></mml:msup></mml:math></inline-formula>. Note that the single-residue model can be seen as a special case of SeqDYN, with <inline-formula><mml:math id="inf104"><mml:msub><mml:mrow><mml:mi>L</mml:mi></mml:mrow><mml:mrow><mml:mi>c</mml:mi><mml:mi>o</mml:mi><mml:mi>r</mml:mi><mml:mi>r</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> set to 0 and <inline-formula><mml:math id="inf105"><mml:mi>q</mml:mi></mml:math></inline-formula> set to <inline-formula><mml:math id="inf106"><mml:mi>m</mml:mi><mml:mi>s</mml:mi><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula>.</p><p>The functional forms of <xref ref-type="disp-formula" rid="equ2">Equation 2a</xref> and <xref ref-type="disp-formula" rid="equ3">Equation 2b</xref> were adapted from <xref ref-type="bibr" rid="bib35">Li et al., 2020</xref>; we also used them for predicting residue-specific membrane association propensities of IDPs (<xref ref-type="bibr" rid="bib52">Qin et al., 2022</xref>). In these previous applications, a linear term was also present in the denominator of <xref ref-type="disp-formula" rid="equ3">Equation 2b</xref>. In our initial training of SeqDYN, the coefficient of the linear term always converged to near zero. We thus eliminated the linear term. In addition to the Lorentzian form, we also tested a Gaussian form for the sequence-distance dependence and found somewhat worse performance. The more gradual attenuation of the Lorentzian form with increasing sequence distance evidently provides an overall better model for the <inline-formula><mml:math id="inf107"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> data in the entire training set. Others (<xref ref-type="bibr" rid="bib11">Cho et al., 2007</xref>; <xref ref-type="bibr" rid="bib63">Sekiyama et al., 2022</xref>; <xref ref-type="bibr" rid="bib18">Delaforge et al., 2018</xref>) have modeled <inline-formula><mml:math id="inf108"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> as the average of some parameters over a window; a window has an extremely abrupt sequence-distance dependence (1 for <inline-formula><mml:math id="inf109"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>s</mml:mi><mml:mo>&lt;</mml:mo><mml:msub><mml:mi>L</mml:mi><mml:mrow><mml:mi>c</mml:mi><mml:mi>o</mml:mi><mml:mi>r</mml:mi><mml:mi>r</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mstyle></mml:math></inline-formula> and 0 for <inline-formula><mml:math id="inf110"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>s</mml:mi><mml:mo>&gt;</mml:mo><mml:msub><mml:mi>L</mml:mi><mml:mrow><mml:mi>c</mml:mi><mml:mi>o</mml:mi><mml:mi>r</mml:mi><mml:mi>r</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mstyle></mml:math></inline-formula>).</p><p>We parametrized the SeqDYN model represented by <xref ref-type="disp-formula" rid="equ2">Equation 2a</xref> and <xref ref-type="disp-formula" rid="equ3">Equation 2b</xref> on the training set of 45 IDPs. In addition to the 21 global parameters noted above, there are also 45 local parameters, namely one uniform scaling factor (<inline-formula><mml:math id="inf111"><mml:mi>Υ</mml:mi></mml:math></inline-formula>) per IDP. The parameter values were selected to minimize the sum of the mean-square-errors for the IDPs in the training set, calculated on <inline-formula><mml:math id="inf112"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> data for a total of 3924 residues. We excluded the 42 Pro residues in the training set because, as already noted, their <inline-formula><mml:math id="inf113"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> values are lower for chemical reasons. We will present validation and test results below, but first let us look at the parameter values.</p><p>The <inline-formula><mml:math id="inf114"><mml:mi>q</mml:mi></mml:math></inline-formula> values are displayed in <xref ref-type="fig" rid="fig3">Figure 3B</xref> alongside <inline-formula><mml:math id="inf115"><mml:mi>m</mml:mi><mml:mi>s</mml:mi><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula>. In descending order, the seven amino acids with the highest <inline-formula><mml:math id="inf116"><mml:mi>q</mml:mi></mml:math></inline-formula> values are Trp, Ile, Tyr, Arg, His, Phe, and Leu. These are exactly the same amino acids in the high-end group for <inline-formula><mml:math id="inf117"><mml:mi>m</mml:mi><mml:mi>s</mml:mi><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula>, though their order there is somewhat different. In ascending order, the seven amino acids (excluding Pro) with the lowest <inline-formula><mml:math id="inf118"><mml:mi>q</mml:mi></mml:math></inline-formula> values are Gly, Asn, Ser, Asp, Val, Thr, and Cys. The composition of the low-end group is also identical to that for <inline-formula><mml:math id="inf119"><mml:mi>m</mml:mi><mml:mi>s</mml:mi><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula>. The <inline-formula><mml:math id="inf120"><mml:mi>q</mml:mi></mml:math></inline-formula> values thus also suggest that π-π and cation-π interactions in local sequences may raise <inline-formula><mml:math id="inf121"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula>, whereas Gly and short-polar residues may lower <inline-formula><mml:math id="inf122"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula>.</p><p>Given the common amino acids at both the high and low ends for <inline-formula><mml:math id="inf123"><mml:mi>m</mml:mi><mml:mi>s</mml:mi><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> and <inline-formula><mml:math id="inf124"><mml:mi>q</mml:mi></mml:math></inline-formula>, it is not surprising that these two properties exhibit a strong correlation, with a coefficient of determination (<italic>R</italic><sup>2</sup>; excluding Pro) at 0.92 (<xref ref-type="fig" rid="fig4">Figure 4A</xref>). Also, because the high-end group contains the largest amino acids (e.g. Trp and Tyr) whereas the low-end group contains the smallest amino acids (e.g. Gly and Ser), we anticipated some correlation of <inline-formula><mml:math id="inf125"><mml:mi>m</mml:mi><mml:mi>s</mml:mi><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> and <inline-formula><mml:math id="inf126"><mml:mi>q</mml:mi></mml:math></inline-formula> with amino-acid size. We measure the latter property by the molecular mass (<italic>m</italic>). As shown in <xref ref-type="fig" rid="fig4">Figure 4B</xref>, both <inline-formula><mml:math id="inf127"><mml:mi>m</mml:mi><mml:mi>s</mml:mi><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> and <inline-formula><mml:math id="inf128"><mml:mi>q</mml:mi></mml:math></inline-formula> indeed show a medium correlation with <italic>m</italic>, with <italic>R</italic><sup>2</sup>=0.67 (excluding Pro) and 0.61, respectively. A bulkiness parameter was proposed as an indicator of sequence-dependent backbone dynamics of IDPs (<xref ref-type="bibr" rid="bib11">Cho et al., 2007</xref>; <xref ref-type="bibr" rid="bib18">Delaforge et al., 2018</xref>). Bulkiness was defined as the sidechain volume-to-length ratio, and identified amino acids with aromatic or branched aliphatic sidechains as bulky (<xref ref-type="bibr" rid="bib83">Zimmerman et al., 1968</xref>). We found only modest correlations between either <inline-formula><mml:math id="inf129"><mml:mi>m</mml:mi><mml:mi>s</mml:mi><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> or <inline-formula><mml:math id="inf130"><mml:mi>q</mml:mi></mml:math></inline-formula> and bulkiness, with <italic>R</italic><sup>2</sup> just below 0.4 (<xref ref-type="fig" rid="fig4">Figure 4C</xref>).</p><fig-group><fig id="fig4" position="float"><label>Figure 4.</label><caption><title>SeqDYN model parameters.</title><p>(<bold>A</bold>) Correlation between <inline-formula><mml:math id="inf131"><mml:mi>m</mml:mi><mml:mi>s</mml:mi><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> and <inline-formula><mml:math id="inf132"><mml:mi>q</mml:mi></mml:math></inline-formula>. The values are also displayed as bars in <xref ref-type="fig" rid="fig3">Figure 3B</xref>. (<bold>B</bold>) Correlation of <inline-formula><mml:math id="inf133"><mml:mi>m</mml:mi><mml:mi>s</mml:mi><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> and <inline-formula><mml:math id="inf134"><mml:mi>q</mml:mi></mml:math></inline-formula> with amino-acid molecular mass. (<bold>C</bold>) Correlation of <inline-formula><mml:math id="inf135"><mml:mi>m</mml:mi><mml:mi>s</mml:mi><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> and <inline-formula><mml:math id="inf136"><mml:mi>q</mml:mi></mml:math></inline-formula> with bulkiness. (<bold>D</bold>) The optimal correlation length and deterioration of SeqDYN prediction as the correlation length is moved away from the optimal value.</p><p><supplementary-material id="fig4sdata1"><label>Figure 4—source data 1.</label><caption><title>Source data for <xref ref-type="fig" rid="fig4">Figure 4</xref>.</title></caption><media mimetype="application" mime-subtype="xlsx" xlink:href="elife-88958-fig4-data1-v1.xlsx"/></supplementary-material></p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-88958-fig4-v1.tif"/></fig><fig id="fig4s1" position="float" specific-use="child-fig"><label>Figure 4—figure supplement 1.</label><caption><title>T-test of on the <inline-formula><mml:math id="inf137"><mml:mi>q</mml:mi></mml:math></inline-formula> parameters for pairs of amino acids.</title><p><inline-formula><mml:math id="inf138"><mml:mi>q</mml:mi></mml:math></inline-formula> parameters were obtained from five-fold cross-validation training, resulting in five independent values for each <inline-formula><mml:math id="inf139"><mml:mi>q</mml:mi></mml:math></inline-formula> parameter. Mean presented as red bars; standard deviation presented as error bar. *, p&lt;0.05; **, p&lt;0.01; ***, p&lt;0.001; ns, not significant. <inline-formula><mml:math id="inf140"><mml:mi>q</mml:mi></mml:math></inline-formula> parameters for all neighboring pairs not explicitly indicated are not significantly different.</p><p><supplementary-material id="fig4s1sdata1"><label>Figure 4—figure supplement 1—source data 1.</label><caption><title>Source data for <xref ref-type="fig" rid="fig4s1">Figure 4—figure supplement 1</xref>.</title></caption><media mimetype="application" mime-subtype="xlsx" xlink:href="elife-88958-fig4-figsupp1-data1-v1.xlsx"/></supplementary-material></p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-88958-fig4-figsupp1-v1.tif"/></fig></fig-group><p>The optimized value of <inline-formula><mml:math id="inf141"><mml:mi>b</mml:mi></mml:math></inline-formula> is 3.164 × 10<sup>–2</sup>, corresponding to an <inline-formula><mml:math id="inf142"><mml:msub><mml:mrow><mml:mi>L</mml:mi></mml:mrow><mml:mrow><mml:mi>c</mml:mi><mml:mi>o</mml:mi><mml:mi>r</mml:mi><mml:mi>r</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> of 5.6 residues. The resulting optimized <inline-formula><mml:math id="inf143"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mrow><mml:mrow><mml:mover><mml:mrow><mml:mi>R</mml:mi><mml:mi>M</mml:mi><mml:mi>S</mml:mi><mml:mi>E</mml:mi></mml:mrow><mml:mo stretchy="false">¯</mml:mo></mml:mover></mml:mrow></mml:mrow></mml:mrow></mml:mstyle></mml:math></inline-formula> is 0.95 s<sup>–1</sup>, a clear improvement over the value 1.24 s<sup>–1</sup> of the null model. To check the sensitivity of prediction accuracy to <inline-formula><mml:math id="inf144"><mml:mi>b</mml:mi></mml:math></inline-formula>, we set <inline-formula><mml:math id="inf145"><mml:mi>b</mml:mi></mml:math></inline-formula> to values corresponding to  <inline-formula><mml:math id="inf146"><mml:msub><mml:mrow><mml:mi>L</mml:mi></mml:mrow><mml:mrow><mml:mi>c</mml:mi><mml:mi>o</mml:mi><mml:mi>r</mml:mi><mml:mi>r</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> = 0, 1, 2,…, and retrained SeqDYN for <inline-formula><mml:math id="inf147"><mml:mi>b</mml:mi></mml:math></inline-formula> fixed at each value (<xref ref-type="fig" rid="fig4">Figure 4D</xref>). Note that the null-model <inline-formula><mml:math id="inf148"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mrow><mml:mrow><mml:mover><mml:mrow><mml:mi>R</mml:mi><mml:mi>M</mml:mi><mml:mi>S</mml:mi><mml:mi>E</mml:mi></mml:mrow><mml:mo stretchy="false">¯</mml:mo></mml:mover></mml:mrow></mml:mrow></mml:mrow></mml:mstyle></mml:math></inline-formula>, 1.24 s<sup>–1</sup>, sets an upper bound. This upper bound is slowly reached when <inline-formula><mml:math id="inf149"><mml:msub><mml:mrow><mml:mi>L</mml:mi></mml:mrow><mml:mrow><mml:mi>c</mml:mi><mml:mi>o</mml:mi><mml:mi>r</mml:mi><mml:mi>r</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> is increased from the optimal value. In the opposite direction, when <inline-formula><mml:math id="inf150"><mml:msub><mml:mrow><mml:mi>L</mml:mi></mml:mrow><mml:mrow><mml:mi>c</mml:mi><mml:mi>o</mml:mi><mml:mi>r</mml:mi><mml:mi>r</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> is decreased from the optimal value, <inline-formula><mml:math id="inf151"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mrow><mml:mrow><mml:mover><mml:mrow><mml:mi>R</mml:mi><mml:mi>M</mml:mi><mml:mi>S</mml:mi><mml:mi>E</mml:mi></mml:mrow><mml:mo stretchy="false">¯</mml:mo></mml:mover></mml:mrow></mml:mrow></mml:mrow></mml:mstyle></mml:math></inline-formula> rises quickly, reaching 1.22 s<sup>–1</sup> at  <inline-formula><mml:math id="inf152"><mml:msub><mml:mrow><mml:mi>L</mml:mi></mml:mrow><mml:mrow><mml:mi>c</mml:mi><mml:mi>o</mml:mi><mml:mi>r</mml:mi><mml:mi>r</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> = 0. The latter <inline-formula><mml:math id="inf153"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mrow><mml:mrow><mml:mover><mml:mrow><mml:mi>R</mml:mi><mml:mi>M</mml:mi><mml:mi>S</mml:mi><mml:mi>E</mml:mi></mml:mrow><mml:mo stretchy="false">¯</mml:mo></mml:mover></mml:mrow></mml:mrow></mml:mrow></mml:mstyle></mml:math></inline-formula> is the same as that of the single-residue model. Lastly we note that there is a strong correlation between the uniform scaling factors and <inline-formula><mml:math id="inf154"><mml:msub><mml:mrow><mml:mover accent="false"><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mo>¯</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> values among the 45 IDPs (<italic>R</italic><sup>2</sup>=0.77), as to be expected. For 39 of the 45 IDPs, <inline-formula><mml:math id="inf155"><mml:mi>Υ</mml:mi></mml:math></inline-formula> values fall in the range of 0.8–2.0 s<sup>–1</sup>.</p><p>As presented next, we evaluate the performance of SeqDYN by leave-one-out cross validation, where each IDP in turn was left out of the training set and the model was trained on the remaining 44 IDPs to predict <inline-formula><mml:math id="inf156"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> for the IDP that was left out. The parameters from the leave-one-out (also known as jackknife) training sessions allow us to assess the potential bias of the training set. For this purpose, we compare the values of the 21 global parameters, either from the full training set or from taking the averages of the jackknife training sessions. For each of the <inline-formula><mml:math id="inf157"><mml:mi>q</mml:mi></mml:math></inline-formula> parameters, the values from these two methods differ only in the fourth digit; for example for Leu, they are both 1.1447 from full training and from jackknife training. The values for <inline-formula><mml:math id="inf158"><mml:mi>b</mml:mi></mml:math></inline-formula> are 3.164×10<sup>–2</sup> from full training as stated above and 3.163×10<sup>–2</sup> from jackknife training. The close agreement in parameter values between full training and jackknife training suggests no significant bias in the training set.</p><p>Another question of interest is whether the difference between the <inline-formula><mml:math id="inf159"><mml:mi>q</mml:mi></mml:math></inline-formula> parameters of two amino acids is statistically significant. To answer this question, we carried out fivefold cross-validation training, resulting in five independent estimates for each parameter. For example, the mean ±standard deviation of the <inline-formula><mml:math id="inf160"><mml:mi>q</mml:mi></mml:math></inline-formula> parameter is 1.1405 ± 0.0066 for Leu and 1.2174 ± 0.0211 for Ile. A t-test shows that their difference is extremely statistically significant (<italic>P</italic>&lt;0.0001). In contrast, the difference between Leu and Phe (<inline-formula><mml:math id="inf161"><mml:mi>q</mml:mi></mml:math></inline-formula>=1.1552 ± 0.0304) is not significant. t-test results for other pairs of amino acids are found in <xref ref-type="fig" rid="fig4s1">Figure 4—figure supplement 1</xref>.</p></sec><sec id="s2-3"><title>Validation of SeqDYN predictions</title><p>We now present leave-one-out cross-validation results. We denote the RMSE of the <inline-formula><mml:math id="inf162"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> prediction for the left-out IDP as RMSE(–1). As expected, RMSE(–1) is higher than the RMSE obtained with the IDP kept in the training set, but the increases are generally slight. Specifically, all but eight of the IDPs have increases &lt;0.1 s<sup>–1</sup>; the largest increase is 0.35 s<sup>–1</sup>, for CBP-ID4. The mean RMSE(–1), or <inline-formula><mml:math id="inf163"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mrow><mml:mrow><mml:mover><mml:mrow><mml:mi>R</mml:mi><mml:mi>M</mml:mi><mml:mi>S</mml:mi><mml:mi>E</mml:mi></mml:mrow><mml:mo stretchy="false">¯</mml:mo></mml:mover></mml:mrow></mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mo>−</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:mstyle></mml:math></inline-formula>, for the 45 IDPs is increased by 0.05 s<sup>–1</sup> over <inline-formula><mml:math id="inf164"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mrow><mml:mrow><mml:mover><mml:mrow><mml:mi>R</mml:mi><mml:mi>M</mml:mi><mml:mi>S</mml:mi><mml:mi>E</mml:mi></mml:mrow><mml:mo stretchy="false">¯</mml:mo></mml:mover></mml:mrow></mml:mrow></mml:mrow></mml:mstyle></mml:math></inline-formula>, to 1.00 s<sup>–1</sup>. The latter value is still a distinct improvement over the mean RMSE 1.24 s<sup>–1</sup> of the null model. The histogram of RMSE(–1) for the 45 IDPs is shown in <xref ref-type="fig" rid="fig5">Figure 5A</xref>. It peaks at 0.5 s<sup>–1</sup>, which is a substantial downshift from the corresponding peak at 0.75 s<sup>–1</sup> for <inline-formula><mml:math id="inf165"><mml:msub><mml:mrow><mml:mi>σ</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:mrow></mml:msub></mml:math></inline-formula> (<xref ref-type="fig" rid="fig3">Figure 3A</xref>). Thirty-four of the 45 IDPs have RMSE(–1) values lower than the corresponding <inline-formula><mml:math id="inf166"><mml:msub><mml:mrow><mml:mi>σ</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:mrow></mml:msub></mml:math></inline-formula>.</p><fig-group><fig id="fig5" position="float"><label>Figure 5.</label><caption><title>Quality of SeqDYN predictions.</title><p>(<bold>A</bold>) Histogram of RMSE(–1). Letters indicate RMSE(–1) values of the IDPs to be presented in panels (<bold>B–F</bold>). (<bold>B–F</bold>) Measured (bars) and predicted (curves) <inline-formula><mml:math id="inf167"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> profiles for MKK4, α-synuclein, Mev-P<sub>NTD</sub>, Sev-NT, and CBP-ID4. In (<bold>E</bold>) and (<bold>F</bold>), green curves are SeqDYN predictions and red curves are obtained after a helix boost.</p><p><supplementary-material id="fig5sdata1"><label>Figure 5—source data 1.</label><caption><title>Source data for <xref ref-type="fig" rid="fig5">Figure 5</xref>.</title></caption><media mimetype="application" mime-subtype="xlsx" xlink:href="elife-88958-fig5-data1-v1.xlsx"/></supplementary-material></p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-88958-fig5-v1.tif"/></fig><fig id="fig5s1" position="float" specific-use="child-fig"><label>Figure 5—figure supplement 1.</label><caption><title>Close reproduction (curve) of the measured <inline-formula><mml:math id="inf168"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> profile (bars) of CBP-ID4 when that set of data alone was used to parameterize SeqDYN.</title><p>The resulting model has no value for predicting <inline-formula><mml:math id="inf169"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> for other proteins.</p><p><supplementary-material id="fig5s1sdata1"><label>Figure 5—figure supplement 1—source data 1.</label><caption><title>Source data for <xref ref-type="fig" rid="fig5s1">Figure 5—figure supplement 1</xref>.</title></caption><media mimetype="application" mime-subtype="xlsx" xlink:href="elife-88958-fig5-figsupp1-data1-v1.xlsx"/></supplementary-material></p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-88958-fig5-figsupp1-v1.tif"/></fig></fig-group><p>To further illustrate the performance of SeqDYN, we present the comparison of predicted and measured <inline-formula><mml:math id="inf170"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> values for five IDPs: MKK4, α-synuclein, Mev-P<sub>NTD</sub>, Sev-NT, and CBP-ID4 (<xref ref-type="fig" rid="fig5">Figure 5B–F</xref>). A simple common feature is the falloff of <inline-formula><mml:math id="inf171"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> at the N- and C-termini, resulting from missing upstream or downstream residues that otherwise would be coupled to the terminal residues, as first recognized by <xref ref-type="bibr" rid="bib61">Schwalbe et al., 1997</xref>. Representative conformations of the five IDPs are displayed in <xref ref-type="fig" rid="fig2">Figure 2</xref>, with residues colored according to the predicted <inline-formula><mml:math id="inf172"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> values. For four of these IDPs, the RMSE(–1) values range from 0.44 to 0.76 s<sup>–1</sup> and are scattered around the peak of the histogram, while the RMSE(–1) for the fifth IDP, namely CBP-ID4, the RMSE(–1) value is 2.01 s<sup>–1</sup> and falls on the tail of the histogram (<xref ref-type="fig" rid="fig5">Figure 5A</xref>). <xref ref-type="fig" rid="fig5">Figure 5B</xref> displays the measured and predicted <inline-formula><mml:math id="inf173"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> for MKK4. SeqDYN correctly predicts higher <inline-formula><mml:math id="inf174"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> values in the second half of the sequence than in the first half. It even correctly predicts the peak around residue Arg75. The sequence in this region is H<sub>72</sub>IERLRTH<sub>79</sub>; six of these eight residues belong to the high-end group. In contrast, the lowest <inline-formula><mml:math id="inf175"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> values occur in the sequence S<sub>7</sub>GGGGSGGGSGSG<sub>19</sub>, comprising entirely of two amino acids in the low-end group.</p><p><inline-formula><mml:math id="inf176"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> values for α-synuclein are shown in <xref ref-type="fig" rid="fig5">Figure 5C</xref>. Here, SeqDYN correctly predicts higher <inline-formula><mml:math id="inf177"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> near the C-terminus and a dip around Gly68. However, it misses the <inline-formula><mml:math id="inf178"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> peaks around Tyr39 and Asp121. MD simulations <xref ref-type="bibr" rid="bib19">Dey et al., 2022</xref> have found that these <inline-formula><mml:math id="inf179"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> peaks can be explained by a combination of secondary structure formation (β-sheet around Tyr39 and polyproline II helix around Asp121) and local (between Tyr39 and Ser42) or long-range (between Asp121 and Lys96) interactions. SeqDYN cannot account for long-range interactions (e.g. between β-strands and between Asp121 and Lys96). <xref ref-type="fig" rid="fig5">Figure 5D</xref> shows that SeqDYN gives excellent <inline-formula><mml:math id="inf180"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> predictions for Mev-P<sub>NTD</sub>. It correctly predicts the high peaks around Arg17, Glu31, Leu193, and lower peaks around Arg235 and Trp285, but does underpredict the narrow peak around Tyr113.</p><p>The overall <inline-formula><mml:math id="inf181"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> profile of Sev-NT is predicted well by SeqDYN, but the peak in the long helical region (residues 478–491) is severely underestimated (green curve in <xref ref-type="fig" rid="fig5">Figure 5E</xref>). A similar situation occurs for CBP-ID4, where the peak in the second long helical region (around Glu113) is underpredicted (green curve in <xref ref-type="fig" rid="fig5">Figure 5F</xref>). While the measured <inline-formula><mml:math id="inf182"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> exhibits a higher peak in the second helical region than in the first helical region (around Arg16), the opposite is predicted by SeqDYN. When the <inline-formula><mml:math id="inf183"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> data were included in the training set (i.e., full training), the second peak is higher than the first one, but that is not a real prediction because the <inline-formula><mml:math id="inf184"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> data themselves were used for training the model. It merely means that the SeqDYN functions can be parameterized to produce any prescribed <inline-formula><mml:math id="inf185"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> profile along the sequence. Indeed, when the <inline-formula><mml:math id="inf186"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> data of CBP-ID4 alone were used to parameterize SeqDYN, the measured <inline-formula><mml:math id="inf187"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> profile is closely reproduced (<xref ref-type="fig" rid="fig5s1">Figure 5—figure supplement 1</xref>). The reversal in <inline-formula><mml:math id="inf188"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> peak heights between the two helical regions is the reason for the aforementioned unusual increase in RMSE when CBP-ID4 was left out of the training set.</p></sec><sec id="s2-4"><title>R2 boost in long helical regions</title><p>It is apparent that SeqDYN underestimates the <inline-formula><mml:math id="inf189"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> of stable long helices. Transient short helices does not seem to be a problem, since these are present, for example in Mev-P<sub>NTD</sub>, where transient helix formation in the first 37 residues and between residues 189–198 (<xref ref-type="bibr" rid="bib43">Milles et al., 2018</xref>) coincides with <inline-formula><mml:math id="inf190"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> peaks that are correctly predicted by SeqDYN. SeqDYN can treat coupling between residues within the correlation length of 5.6 residues, but a much longer helix would tumble more slowly than implied by an <inline-formula><mml:math id="inf191"><mml:msub><mml:mrow><mml:mi>L</mml:mi></mml:mrow><mml:mrow><mml:mi>c</mml:mi><mml:mi>o</mml:mi><mml:mi>r</mml:mi><mml:mi>r</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> of 5.6, and thus it makes sense that SeqDYN would underestimate <inline-formula><mml:math id="inf192"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> in that case.</p><p>Our solution then is to apply a boost factor to the long helical region. To do so, we have to know whether an IDP does form long helices and if so what the constituent residues are. Secondary structure predictors tend to overpredict α-helices and β-strands for IDPs, as they are trained on structured proteins. One way to counter that tendency is to make the criteria for α-helices and β-strands stricter. We found that, by filtering PsiPred (<ext-link ext-link-type="uri" xlink:href="http://bioinf.cs.ucl.ac.uk/psipred">http://bioinf.cs.ucl.ac.uk/psipred</ext-link>; <xref ref-type="bibr" rid="bib42">McGuffin et al., 2000</xref>) helix propensity scores (<inline-formula><mml:math id="inf193"><mml:msub><mml:mrow><mml:mi>p</mml:mi></mml:mrow><mml:mrow><mml:mi>H</mml:mi><mml:mi>l</mml:mi><mml:mi>x</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>) with a very high cutoff of 0.99, the surviving helix predictions usually correspond well with residues identified by NMR as having high helix propensities. For example, for Mev-P<sub>NTD</sub>, PsiPred plus filtering predicts residues 14–17, 28–33, and 191–193 as helical; all of them are in regions that form transient helices according to chemical shifts (<xref ref-type="bibr" rid="bib43">Milles et al., 2018</xref>). Likewise long helices are also correctly predicted for Sev-NT (residues 477–489) and CBP-ID4 (residues 6–17 and 105–116; <xref ref-type="bibr" rid="bib29">Jensen et al., 2008</xref>; <xref ref-type="bibr" rid="bib51">Piai et al., 2016</xref>).</p><p>We apply a boost factor, <inline-formula><mml:math id="inf194"><mml:msub><mml:mrow><mml:mi>B</mml:mi></mml:mrow><mml:mrow><mml:mi>H</mml:mi><mml:mi>l</mml:mi><mml:mi>x</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>, to helices with a threshold length of 12:<disp-formula id="equ4"><label>(3)</label><mml:math id="m4"><mml:mrow><mml:msub><mml:mi>B</mml:mi><mml:mrow><mml:mi>H</mml:mi><mml:mi>l</mml:mi><mml:mi>x</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mn>1</mml:mn><mml:mo>+</mml:mo><mml:mi>α</mml:mi><mml:msub><mml:mi>p</mml:mi><mml:mrow><mml:mi>H</mml:mi><mml:mi>l</mml:mi><mml:mi>x</mml:mi></mml:mrow></mml:msub><mml:mi mathvariant="normal">Θ</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mi>p</mml:mi><mml:mrow><mml:mi>H</mml:mi><mml:mi>l</mml:mi><mml:mi>x</mml:mi></mml:mrow></mml:msub><mml:mo>≥</mml:mo><mml:mn>0.99</mml:mn><mml:mo>;</mml:mo><mml:msub><mml:mi>L</mml:mi><mml:mrow><mml:mi>H</mml:mi><mml:mi>l</mml:mi><mml:mi>x</mml:mi></mml:mrow></mml:msub><mml:mo>≥</mml:mo><mml:mn>12</mml:mn></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:math></disp-formula></p><p>The <inline-formula><mml:math id="inf195"><mml:mi>Θ</mml:mi></mml:math></inline-formula> function is 1 if the helix propensity score is above the filtering cutoff and the helix length (<inline-formula><mml:math id="inf196"><mml:msub><mml:mrow><mml:mi>L</mml:mi></mml:mrow><mml:mrow><mml:mi>H</mml:mi><mml:mi>l</mml:mi><mml:mi>x</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>) is above the threshold, and 0 otherwise With a boost amplitude <inline-formula><mml:math id="inf197"><mml:mi>α</mml:mi></mml:math></inline-formula> at 0.5, the boosted SeqDYN prediction for Sev-NT reaches excellent agreement with the measured <inline-formula><mml:math id="inf198"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> (<xref ref-type="fig" rid="fig5">Figure 5E</xref>, red curve). The RMSE(–1) is reduced from 0.76 s<sup>–1</sup> to 0.38 s<sup>–1</sup> upon boosting. Applying the same helix boost to CBP-ID4 also results in a modest reduction in RMSE(–1), from 2.01 to 1.90 s<sup>–1</sup> (<xref ref-type="fig" rid="fig5">Figure 5F</xref>, red curve). The only other IDP for which PsiPred plus filtering predicts a long helix is the N-terminal region of lysyl-tRNA synthetase (KRS-NT). The authors who studied this protein did not report on secondary structure (<xref ref-type="bibr" rid="bib12">Cho et al., 2014</xref>), but feeding their reported chemical shifts to the TALOS +server (<ext-link ext-link-type="uri" xlink:href="https://spin.niddk.nih.gov/bax/nmrserver/talos/">https://spin.niddk.nih.gov/bax/nmrserver/talos/</ext-link>; <xref ref-type="bibr" rid="bib64">Shen et al., 2009</xref>) found only short stretches of residues that fall into the helical region of the Ramachandran map. The SeqDYN prediction for KRS-NT is already good [RMSE(–1)=0.83 s<sup>–1</sup>]; applying a helix boost would deteriorate the RMSE(−1) to 1.16 s<sup>–1</sup>.</p></sec><sec id="s2-5"><title>Further test on a set of nine IDPs</title><p>We have reserved nine IDPs for testing SeqDYN (parameterized on the training set of 45 IDPs). The level of disorder in these test proteins also spans the full range, from absence of secondary structures [ChiZ N-terminal region (<xref ref-type="bibr" rid="bib25">Hicks et al., 2020</xref>), Pdx1 C-terminal region (<xref ref-type="bibr" rid="bib15">Cook et al., 2019</xref>), and TIA-1 prion-like domain (<xref ref-type="bibr" rid="bib63">Sekiyama et al., 2022</xref>)] to presence of transient short helices [synaptobrevin-2 (<xref ref-type="bibr" rid="bib32">Lakomek et al., 2019</xref>), α-endosulfine (<xref ref-type="bibr" rid="bib71">Thapa et al., 2022</xref>), YAP (<xref ref-type="bibr" rid="bib21">Feichtinger et al., 2022</xref>), angiomotin-like 1 (AMOTL1) (<xref ref-type="bibr" rid="bib73">Vogel et al., 2022</xref>)] to formation of stable long helices [FtsQ <xref ref-type="bibr" rid="bib65">Smrt et al., 2023</xref> and CAHS-8 <xref ref-type="bibr" rid="bib38">Malki et al., 2022</xref>]. For eight of the nine test IDPs, the RMSEs of SeqDYN predictions are lower than the experimental <inline-formula><mml:math id="inf199"><mml:msub><mml:mrow><mml:mi>σ</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:mrow></mml:msub></mml:math></inline-formula> values, by an average of 0.66 s<sup>–1</sup>. For the ninth IDP (Pdx1), the SeqDYN RMSE is slightly higher, by 0.06 s<sup>–1</sup>, than the experimental <inline-formula><mml:math id="inf200"><mml:msub><mml:mrow><mml:mi>σ</mml:mi></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:mrow></mml:msub></mml:math></inline-formula>. Together, the nine test IDPs have a mean RMSE of 1.13 s<sup>–1</sup>, close to the <inline-formula><mml:math id="inf201"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mrow><mml:mrow><mml:mover><mml:mrow><mml:mi>R</mml:mi><mml:mi>M</mml:mi><mml:mi>S</mml:mi><mml:mi>E</mml:mi></mml:mrow><mml:mo stretchy="false">¯</mml:mo></mml:mover></mml:mrow></mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mo>−</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:mstyle></mml:math></inline-formula> of 1.00 s<sup>–1</sup> for the training set in the leave-one-out cross-validation.</p><p>The comparison of predicted and measured <inline-formula><mml:math id="inf202"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> profiles along the sequence is presented in <xref ref-type="fig" rid="fig6">Figure 6A–I</xref>. For ChiZ, SeqDYN correctly predicts the major peak around Arg25 and the minor peak around Arg46 (<xref ref-type="fig" rid="fig6">Figure 6A</xref>). The <inline-formula><mml:math id="inf203"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> profile of Pdx1 is largely featureless, except for a dip around Gly216, which is correctly predicted by SeqDYN (<xref ref-type="fig" rid="fig6">Figure 6B</xref>). Correct prediction is also obtained for the higher <inline-formula><mml:math id="inf204"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> in the first half of TIA-1 prion-like domain than in the second half (<xref ref-type="fig" rid="fig6">Figure 6C</xref>). SeqDYN gives an excellent prediction for synaptobrevin-2, including a linear increase up to Arg56 and the major peak around Trp89 (<xref ref-type="fig" rid="fig6">Figure 6D</xref>).</p><fig id="fig6" position="float"><label>Figure 6.</label><caption><title>Measured (bars) and predicted (curves) <inline-formula><mml:math id="inf205"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> profiles for ChiZ N-terminal region, TIA1 prion-like domain, Pdx1 C-terminal region, synaptobrevin-2, α-endosulfine, YAP, AMOTL1, FtsQ, and CAHS-8.</title><p>In (<bold>C</bold>), <inline-formula><mml:math id="inf206"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> does not fall off at the N-terminus because the sequence is preceded by an expression tag MGSSHHHHHHHHHHHHS. In (<bold>H</bold>) and (<bold>I</bold>), green curves are SeqDYN predictions and red curves are obtained after a helix boost.</p><p><supplementary-material id="fig6sdata1"><label>Figure 6—source data 1.</label><caption><title>Source data for <xref ref-type="fig" rid="fig6">Figure 6</xref>.</title></caption><media mimetype="application" mime-subtype="xlsx" xlink:href="elife-88958-fig6-data1-v1.xlsx"/></supplementary-material></p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-88958-fig6-v1.tif"/></fig><p>The prediction is also very good for α-endosulfine, including elevated <inline-formula><mml:math id="inf207"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> around Glu34, which coincides with the presence of a transient helix, and depressed <inline-formula><mml:math id="inf208"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> in the last 40 residues (<xref ref-type="fig" rid="fig6">Figure 6E</xref>). The only miss is an underprediction for the peak around Lys74. SeqDYN also predicts well the overall shape of the <inline-formula><mml:math id="inf209"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> profile for YAP, including peaks around Asn70, Leu91, Arg124, and Arg161, but severely underestimates the peak height around Asn70 (<xref ref-type="fig" rid="fig6">Figure 6F</xref>). NOE signals indicate contacts between Met86, Leu91, Fhe95, and Fhe96 (<xref ref-type="bibr" rid="bib21">Feichtinger et al., 2022</xref>); evidently this type of local contacts is captured well by SeqDYN. The <inline-formula><mml:math id="inf210"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> elevation around Asn70 is mostly due to helix formation: residues 61–74 have helix propensities up to 40% (<xref ref-type="bibr" rid="bib21">Feichtinger et al., 2022</xref>). PsiPred predicts helix for residues 62–73, but only residues 65–68 survive the filtering that we impose, resulting in a helix that is too short to apply a helix boost. The prediction for AMOTL1 is mostly satisfactory, including peaks around Phe200 and Arg264 and a significant dip around Gly292 (<xref ref-type="fig" rid="fig6">Figure 6G</xref>). However, whereas the two peaks have approximately equal heights in the measured <inline-formula><mml:math id="inf211"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> profile, the predicted peak height around Phe200 is too low. SCSs indicate helix propensity around both <inline-formula><mml:math id="inf212"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> peaks (<xref ref-type="bibr" rid="bib73">Vogel et al., 2022</xref>). PsiPred also predicts helix in both regions, but only five and two residues, respectively, survive after filtering, and are too short for applying a helix boost.</p><p>For FtsQ, SeqDYN correctly predicts elevated <inline-formula><mml:math id="inf213"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> for the long helix [residues 46–74 <xref ref-type="bibr" rid="bib65">Smrt et al., 2023</xref>] but underestimates the magnitude (RMSE = 2.32 s<sup>–1</sup>; green curve in <xref ref-type="fig" rid="fig6">Figure 6H</xref>). PsiPred plus filtering predicts a long helix formed by residues 47–73. Applying the helix boost substantially improves the agreement with the measured <inline-formula><mml:math id="inf214"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula>, with RMSE reducing to 1.71 s<sup>–1</sup> (red curve in <xref ref-type="fig" rid="fig6">Figure 6H</xref>). SeqDYN also gives a qualitatively correct <inline-formula><mml:math id="inf215"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> profile for CAHS-8, with higher <inline-formula><mml:math id="inf216"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> for the middle section (residues 95–190; RMSE = 2.36 s<sup>–1</sup>; green curve in <xref ref-type="fig" rid="fig6">Figure 6I</xref>). However, it misses the extra elevation in <inline-formula><mml:math id="inf217"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> for the first half of the middle section (residues 95–145). According to SCS, the first and second halves have helix propensities of 60% and 30%, respectively (<xref ref-type="bibr" rid="bib38">Malki et al., 2022</xref>). PsiPred plus filtering predicts helices for residues 96–121, 124–141, 169–173, and 179–189. Only the first two helices, both in the first half of the middle section, are considered long according to our threshold. Once again, applying the helix boost leads to marked improvement in the predicted in <inline-formula><mml:math id="inf218"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula>, with RMSE reducing to 1.92 s<sup>–1</sup> (red curve in <xref ref-type="fig" rid="fig6">Figure 6I</xref>).</p></sec><sec id="s2-6"><title>Inputting the sequences of structured proteins predicts <inline-formula><mml:math id="inf219"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> in the unfolded state</title><p>SeqDYN is trained on IDPs, what if we feed it with the sequence of a structured protein? The prediction using the sequence of hen egg white lysozyme, a well-studied single-domain protein, is displayed in <xref ref-type="fig" rid="fig7">Figure 7A</xref>. It shows remarkable agreement with the <inline-formula><mml:math id="inf220"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> profile measured by Klein-<xref ref-type="bibr" rid="bib31">Klein-Seetharaman et al., 2002</xref> in the unfolded state (denatured by 8 M urea at pH 2 and reduced to break disulfide bridges), including a major peak around Trp62, a second peak around Trp111, and a third peak around Trp123. Klein-Seetharaman et al. mutated Trp62 to Gly and the major peak all but disappeared. This result is also precisely predicted by SeqDYN with the mutant sequence (<xref ref-type="fig" rid="fig7">Figure 7B</xref>).</p><fig id="fig7" position="float"><label>Figure 7.</label><caption><title><inline-formula><mml:math id="inf221"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> profiles predicted (curves) by SeqDYN show close agreement with those measured (bars) on structured proteins in the unfolded state.</title><p>(<bold>A</bold>) Wild-type lysozyme (8 M urea; pH 2; cysteine-methylated). (<bold>B</bold>) Lysozyme with Trp62 to Gly mutation (pH 2). Methylated cysteines were treated as Ala in the SeqDYN predictions. (<bold>C</bold>) Apomyoglogin (8 M urea; pH 2.3). (<bold>D</bold>) Ubiquitin (8 M urea; pH 2).</p><p><supplementary-material id="fig7sdata1"><label>Figure 7—source data 1.</label><caption><title>Source data for <xref ref-type="fig" rid="fig7">Figure 7</xref>.</title></caption><media mimetype="application" mime-subtype="xlsx" xlink:href="elife-88958-fig7-data1-v1.xlsx"/></supplementary-material></p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-88958-fig7-v1.tif"/></fig><p>SeqDYN also predicts well the <inline-formula><mml:math id="inf222"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> profiles of other proteins in the unfolded state. For unfolded apomyoglobin (8 M urea; pH 2.3), <xref ref-type="bibr" rid="bib62">Schwarzinger et al., 2002</xref> claimed that depressed <inline-formula><mml:math id="inf223"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> corresponded to stretches of small amino acids (Gly and Ala), whereas elevated corresponded to local hydrophobic interactions. SeqDYN reproduces all the observed peaks and valleys in the <inline-formula><mml:math id="inf224"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> profile (<xref ref-type="fig" rid="fig7">Figure 7C</xref>). The deepest valley indeed occurs over a Gly/Ala-rich stretch, G<sub>125</sub>ADAQGA<underline><sub>131</sub></underline>, but the highest peak occurs over a stretch, I<sub>102</sub>KYLEFI<sub>108</sub>, that contains both hydrophobic and charged residues, all of which are on the high end of the <inline-formula><mml:math id="inf225"><mml:mi>q</mml:mi></mml:math></inline-formula> parameters (<xref ref-type="fig" rid="fig3">Figure 3B</xref>). The <inline-formula><mml:math id="inf226"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> profile of unfolded ubiquitin (8 M urea; pH 2) is relatively flat, which <xref ref-type="bibr" rid="bib76">Wirmer et al., 2006</xref> attributed to lack of residual secondary structure, based on the assumption that β-sheets (major elements of folded ubiquitin) are less resistant to denaturation than α-helices. SeqDYN predicts a relatively flat <inline-formula><mml:math id="inf227"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> profile (<xref ref-type="fig" rid="fig7">Figure 7D</xref>), but the reason is that the ubiquitin sequence lacks a contiguous stretch of high-<inline-formula><mml:math id="inf228"><mml:mi>q</mml:mi></mml:math></inline-formula> amino acids.</p></sec></sec><sec id="s3" sec-type="discussion"><title>Discussion</title><p>We have developed a powerful method, SeqDYN, that predicts the backbone amide transverse relaxation rates (<inline-formula><mml:math id="inf229"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula>) of IDPs. The method is based on IDP sequences, is extremely fast, and available as a web server at <ext-link ext-link-type="uri" xlink:href="https://zhougroup-uic.github.io/SeqDYNidp/">https://zhougroup-uic.github.io/SeqDYNidp/</ext-link> (<xref ref-type="bibr" rid="bib53">Qin and Zhou, 2024</xref>). The excellent performance supports the notion that the ns-dynamics reported by <inline-formula><mml:math id="inf230"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> is coded by the local sequence, comprising up to 6 residues on either side of a given residue. The amino-acid types that contribute the most to coupling within a local sequence are aromatic (Trp, Tyr, Phe, and His), Arg, and long branched aliphatic (Ile and Leu), suggesting the importance of π-π, cation-π, and hydrophobic interactions in raising <inline-formula><mml:math id="inf231"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula>. These interactions are interrupted by Gly and amino acids with short polar sidechains (Ser, Thr, Asn, and Asp), leading to reduced <inline-formula><mml:math id="inf232"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula>. Transient short helices produce moderate elevation in <inline-formula><mml:math id="inf233"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula>, whereas stable long helices result in a big boost in <inline-formula><mml:math id="inf234"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula>. Tertiary contacts can also raise <inline-formula><mml:math id="inf235"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula>, but appear to be infrequent in most IDPs (<xref ref-type="bibr" rid="bib19">Dey et al., 2022</xref>).</p><p>It is also possible that <inline-formula><mml:math id="inf236"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> reported by backbone amide <sup>15</sup>N relaxation (as is the case for most of the IDPs studied here) may not be particularly sensitive to exchange effects, which likely involve tertiary contact formation. For the D2 domain of p27<sup>Kip1</sup>, the exchange contributions measured using <sup>15</sup>N relaxation were small (&lt;2.5 s<sup>–1</sup>) but were as large as 25 s<sup>–1</sup> when measured by high-power <sup>1</sup>H relaxation dispersion (<xref ref-type="bibr" rid="bib4">Ban et al., 2017</xref>). This experiment measures the effective transverse relation rate, <inline-formula><mml:math id="inf237"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn><mml:mo>,</mml:mo><mml:mi>e</mml:mi><mml:mi>f</mml:mi><mml:mi>f</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>, over a range of effective radiofrequency <inline-formula><mml:math id="inf238"><mml:msub><mml:mrow><mml:mi>ω</mml:mi></mml:mrow><mml:mrow><mml:mi>e</mml:mi><mml:mi>f</mml:mi><mml:mi>f</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>. The exchange contribution is maximal for the value <inline-formula><mml:math id="inf239"><mml:msubsup><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn><mml:mo>,</mml:mo><mml:mi>e</mml:mi><mml:mi>f</mml:mi><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi><mml:mi>o</mml:mi><mml:mi>w</mml:mi><mml:msub><mml:mrow><mml:mi>ω</mml:mi></mml:mrow><mml:mrow><mml:mi>e</mml:mi><mml:mi>f</mml:mi><mml:mi>f</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msubsup></mml:math></inline-formula> at low <inline-formula><mml:math id="inf240"><mml:msub><mml:mrow><mml:mi>ω</mml:mi></mml:mrow><mml:mrow><mml:mi>e</mml:mi><mml:mi>f</mml:mi><mml:mi>f</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> but is largely quenched for the value <inline-formula><mml:math id="inf241"><mml:msubsup><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2,0</mml:mn></mml:mrow><mml:mrow><mml:mi>a</mml:mi><mml:mi>p</mml:mi><mml:mi>p</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula> in the high-<inline-formula><mml:math id="inf242"><mml:msub><mml:mrow><mml:mi>ω</mml:mi></mml:mrow><mml:mrow><mml:mi>e</mml:mi><mml:mi>f</mml:mi><mml:mi>f</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> limit. The SeqDYN prediction for this IDP matches much better with <inline-formula><mml:math id="inf243"><mml:msubsup><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2,0</mml:mn></mml:mrow><mml:mrow><mml:mi>a</mml:mi><mml:mi>p</mml:mi><mml:mi>p</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula> than with <inline-formula><mml:math id="inf244"><mml:msubsup><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn><mml:mo>,</mml:mo><mml:mi>e</mml:mi><mml:mi>f</mml:mi><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mi>l</mml:mi><mml:mi>o</mml:mi><mml:mi>w</mml:mi><mml:msub><mml:mrow><mml:mi>ω</mml:mi></mml:mrow><mml:mrow><mml:mi>e</mml:mi><mml:mi>f</mml:mi><mml:mi>f</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msubsup></mml:math></inline-formula> (<xref ref-type="fig" rid="fig8">Figure 8</xref>). It is not clear whether this IDP is unique in forming persistent tertiary contacts that give rise to substantial exchange contributions or the <sup>1</sup>H relaxation dispersion experiment is unique in reporting the exchange contributions. At the minimum, SeqDYN yields the exchange-free portion of the transverse relaxation rate, enabling easy identification of residues that potentially participate in tertiary contacts. For the D2 domain of p27<sup>Kip1</sup>, SeqDYN correctly predicts the <inline-formula><mml:math id="inf245"><mml:msubsup><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2,0</mml:mn></mml:mrow><mml:mrow><mml:mi>a</mml:mi><mml:mi>p</mml:mi><mml:mi>p</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula> local maxima at W76 and Y88. It is these same two residues that show substantial exchange contributions and putatively participate in tertiary contact (<xref ref-type="bibr" rid="bib4">Ban et al., 2017</xref>). Therefore local contacts may seed tertiary contacts. If <inline-formula><mml:math id="inf246"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn><mml:mo>,</mml:mo><mml:mi>e</mml:mi><mml:mi>f</mml:mi><mml:mi>f</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> data with substantial exchange contributions become available for more IDPs, SeqDYN may be retrained to make predictions for IDPs forming persistent tertiary contacts.</p><fig id="fig8" position="float"><label>Figure 8.</label><caption><title>Comparison between SeqDYN prediction (curves) and effective transverse relaxation rate (bars) from <sup>1</sup>H dispersion relaxation experiment.</title><p>(<bold>A</bold>) <inline-formula><mml:math id="inf247"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn><mml:mo>,</mml:mo><mml:mi>e</mml:mi><mml:mi>f</mml:mi><mml:mi>f</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> in the high-<inline-formula><mml:math id="inf248"><mml:msub><mml:mrow><mml:mi>ω</mml:mi></mml:mrow><mml:mrow><mml:mi>e</mml:mi><mml:mi>f</mml:mi><mml:mi>f</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> limit. (<bold>B</bold>) <inline-formula><mml:math id="inf249"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn><mml:mo>,</mml:mo><mml:mi>e</mml:mi><mml:mi>f</mml:mi><mml:mi>f</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> at low <inline-formula><mml:math id="inf250"><mml:msub><mml:mrow><mml:mi>ω</mml:mi></mml:mrow><mml:mrow><mml:mi>e</mml:mi><mml:mi>f</mml:mi><mml:mi>f</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>.</p><p><supplementary-material id="fig8sdata1"><label>Figure 8—source data 1.</label><caption><title>Source data for <xref ref-type="fig" rid="fig8">Figure 8</xref>.</title></caption><media mimetype="application" mime-subtype="xlsx" xlink:href="elife-88958-fig8-data1-v1.xlsx"/></supplementary-material></p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-88958-fig8-v1.tif"/></fig><p>The <inline-formula><mml:math id="inf251"><mml:mi>q</mml:mi></mml:math></inline-formula> parameters, while introduced here to characterize the propensities of amino acids to participate in local interactions, appear to correlate with the tendencies of amino acids to drive liquid-liquid phase separation. Consistent with the rank order of <inline-formula><mml:math id="inf252"><mml:mi>q</mml:mi></mml:math></inline-formula>, Trp, Tyr, and Arg have been reported to be strong drivers of phase separation, Lys is a moderate driver, whereas Gly and Ser suppress phase separation (<xref ref-type="bibr" rid="bib40">Martin et al., 2020</xref>; <xref ref-type="bibr" rid="bib77">Wong et al., 2020</xref>; <xref ref-type="bibr" rid="bib75">Wang et al., 2018</xref>). Recent measurements of the threshold concentration produced the following order for the propensity of phase separation by eight nonpolar amino acids in homotetrapeptides of the form XXssXX (ss: backbone disulfide bond): Trp &gt; Phe &gt; Leu&gt;Met &gt; Ile&gt;Val &gt; Ala&gt;Pro (<xref ref-type="bibr" rid="bib81">Zhang et al., 2024</xref>). This order is the same as that of the <inline-formula><mml:math id="inf253"><mml:mi>q</mml:mi></mml:math></inline-formula> parameters, except that the <inline-formula><mml:math id="inf254"><mml:mi>q</mml:mi></mml:math></inline-formula> values of Ile and Val are in the second and last places, respectively. Threshold concentrations of IDPs are now predicted reasonably well by coarse-grained simulations where each amino acid is modeled by a single bead with a Lennard-Jones diameter <inline-formula><mml:math id="inf255"><mml:msub><mml:mrow><mml:mi>d</mml:mi></mml:mrow><mml:mrow><mml:mn>0</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> and a stickiness parameter <inline-formula><mml:math id="inf256"><mml:mi>λ</mml:mi></mml:math></inline-formula> (<xref ref-type="bibr" rid="bib70">Tesei and Lindorff-Larsen, 2022</xref>). Our <inline-formula><mml:math id="inf257"><mml:mi>q</mml:mi></mml:math></inline-formula> parameter shows a good correlation (<italic>R</italic><sup>2</sup>=0.59) with the compound parameter <inline-formula><mml:math id="inf258"><mml:msubsup><mml:mrow><mml:mi>d</mml:mi></mml:mrow><mml:mrow><mml:mn>0</mml:mn></mml:mrow><mml:mrow><mml:mn>3</mml:mn></mml:mrow></mml:msubsup><mml:mi>λ</mml:mi></mml:math></inline-formula> (<xref ref-type="fig" rid="fig9">Figure 9</xref>). Therefore, the <inline-formula><mml:math id="inf259"><mml:mi>q</mml:mi></mml:math></inline-formula> parameter may serve as a predictor for the tendency of an amino acid to drive phase separation. In essence, the same ability of an amino acid, for example Trp, to form interactions with neighboring residues of an IDP in the free state also applies when it comes to interactions with residues on neighboring chains in a dense phase.</p><fig id="fig9" position="float"><label>Figure 9.</label><caption><title>Correlation between the stickiness parameters (<italic>λ</italic>) and the NMR relaxation parameters (q).</title><p>The regression line is shown as dashes.</p><p><supplementary-material id="fig9sdata1"><label>Figure 9—source data 1.</label><caption><title>Source data for <xref ref-type="fig" rid="fig9">Figure 9</xref>.</title></caption><media mimetype="application" mime-subtype="xlsx" xlink:href="elife-88958-fig9-data1-v1.xlsx"/></supplementary-material></p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-88958-fig9-v1.tif"/></fig><p>Our method incorporates ideas from a number of previous efforts at describing <inline-formula><mml:math id="inf260"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula>. The first serious effort was by <xref ref-type="bibr" rid="bib61">Schwalbe et al., 1997</xref>, who accounted for contributions from neighboring residues as additive terms, instead of multiplicative factors as in SeqDYN. <xref ref-type="bibr" rid="bib11">Cho et al., 2007</xref> and <xref ref-type="bibr" rid="bib18">Delaforge et al., 2018</xref> used the running average of the bulkiness parameter over a window of five to nine residues as a qualitative indicator of <inline-formula><mml:math id="inf261"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula>. Here again the calculation was based on an additive model. <xref ref-type="bibr" rid="bib63">Sekiyama et al., 2022</xref> employed a multiplicative model, with <inline-formula><mml:math id="inf262"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> calculated as a geometric mean of ‘indices of local dynamics’ over a five-residue window. These indices, akin to our <inline-formula><mml:math id="inf263"><mml:mi>q</mml:mi></mml:math></inline-formula> parameters, were trained on a single IDR (TIA-1 prion-like domain) and used to reproduce the measured <inline-formula><mml:math id="inf264"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> for the same IDR. As we have illustrated on CBP-ID4 (<xref ref-type="fig" rid="fig5s1">Figure 5—figure supplement 1</xref>), training on a single protein merely biases the parameters to that model and has little value in predicting <inline-formula><mml:math id="inf265"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> for other proteins. In comparison, SeqDYN is trained on 45 IDPs and its predictions are robust and achieve quantitative agreement with measured <inline-formula><mml:math id="inf266"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula>.</p><p>Ten of the IDPs tested here have been studied recently by MD simulations using IDP-specific force fields (<xref ref-type="bibr" rid="bib19">Dey et al., 2022</xref>; <xref ref-type="bibr" rid="bib25">Hicks et al., 2020</xref>; <xref ref-type="bibr" rid="bib80">Yu and Brüschweiler, 2022</xref>; <xref ref-type="bibr" rid="bib65">Smrt et al., 2023</xref>). In <xref ref-type="table" rid="table2">Table 2</xref>, we compare the RMSEs of SeqDYN predictions with those for <inline-formula><mml:math id="inf267"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> calculations from MD simulations. For five of these IDPs: A1-LCD, Aβ40, α-synuclein, tau K18, and FtsQ, RMSEs of SeqDYN and MD are remarkably similar. Four of these IDPs lack significant population of α-helices or β-sheets, but FtsQ forms a stable long helix. For one other IDP, namely HOX-DFD, MD, by explicitly modeling its folded domain, does a much better job in predicting <inline-formula><mml:math id="inf268"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> than SeqDYN (RMSEs of 1.40 s<sup>–1</sup> vs 1.99 s<sup>–1</sup>). However, for the four remaining IDPs: p53TAD, Pup, Sev-NT, and ChiZ, SeqDYN significantly outperforms MD, with RMSEs averaging only 0.47 s<sup>–1</sup>, compared to the MD counterpart of 1.14 s<sup>–1</sup>. Overall, SeqDYN is very competitive against MD in predicting <inline-formula><mml:math id="inf269"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula>, but without the significant computational cost. While MD simulations can reveal details of local interactions, as noted for α-synuclein, and capture tertiary interactions if they occur, they still suffer from perennial problems of force-field imperfection and inadequate sampling. SeqDYN provides an accurate description of IDP dynamics at a ‘mean-field’ level, but could miss idiosyncratic behaviors of specific local sequences.</p><table-wrap id="table2" position="float"><label>Table 2.</label><caption><title>RMSEs (s<sup>–1</sup>) of <inline-formula><mml:math id="inf270"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> predictions by SeqDYN and MD for 10 IDPs.</title></caption><table frame="hsides" rules="groups"><thead><tr><th align="left" valign="bottom">IDP name</th><th align="left" valign="bottom">SeqDYN</th><th align="left" valign="bottom">MD</th></tr></thead><tbody><tr><td align="left" valign="bottom">A1-LCD</td><td align="char" char="." valign="bottom">0.60<xref ref-type="table-fn" rid="table2fn1">*</xref></td><td align="char" char="." valign="bottom">0.59 <xref ref-type="table-fn" rid="table2fn4"><sup>§</sup></xref><sup>, <xref ref-type="table-fn" rid="table2fn5">¶</xref></sup></td></tr><tr><td align="left" valign="bottom">Aβ40</td><td align="char" char="." valign="bottom">0.38<xref ref-type="table-fn" rid="table2fn1">*</xref></td><td align="char" char="." valign="bottom">0.38 <xref ref-type="table-fn" rid="table2fn4"><sup>§</sup></xref></td></tr><tr><td align="left" valign="bottom">HOX-DFD</td><td align="char" char="." valign="bottom">1.99<xref ref-type="table-fn" rid="table2fn1">*</xref></td><td align="char" char="." valign="bottom">1.40 <xref ref-type="table-fn" rid="table2fn4"><sup>§</sup></xref></td></tr><tr><td align="left" valign="bottom">α-synuclein</td><td align="char" char="." valign="bottom">0.44<xref ref-type="table-fn" rid="table2fn1">*</xref></td><td align="char" char="." valign="bottom">0.50 <xref ref-type="table-fn" rid="table2fn4"><sup>§</sup></xref></td></tr><tr><td align="left" valign="bottom">p53TAD</td><td align="char" char="." valign="bottom">0.33<xref ref-type="table-fn" rid="table2fn1">*</xref></td><td align="char" char="." valign="bottom">1.04 <sup><xref ref-type="table-fn" rid="table2fn6">**</xref></sup></td></tr><tr><td align="left" valign="bottom">Pup</td><td align="char" char="." valign="bottom">0.43<xref ref-type="table-fn" rid="table2fn1">*</xref></td><td align="char" char="." valign="bottom">1.00<xref ref-type="table-fn" rid="table2fn6">**</xref></td></tr><tr><td align="left" valign="bottom">Sev-NT</td><td align="char" char="." valign="bottom">0.38<xref ref-type="table-fn" rid="table2fn1">*</xref><sup>,<xref ref-type="table-fn" rid="table2fn2">†</xref></sup></td><td align="char" char="." valign="bottom">1.10 <xref ref-type="table-fn" rid="table2fn4"><sup>§</sup></xref><sup>,<xref ref-type="table-fn" rid="table2fn7">††</xref></sup></td></tr><tr><td align="left" valign="bottom">tau K18</td><td align="char" char="." valign="bottom">0.83<xref ref-type="table-fn" rid="table2fn1">*</xref></td><td align="char" char="." valign="bottom">0.80 <xref ref-type="table-fn" rid="table2fn4"><sup>§</sup></xref></td></tr><tr><td align="left" valign="bottom">ChiZ</td><td align="char" char="." valign="bottom">0.74 <xref ref-type="table-fn" rid="table2fn3"><sup>‡</sup></xref></td><td align="char" char="." valign="bottom">1.40 <xref ref-type="table-fn" rid="table2fn8"><sup>‡ ‡</sup></xref></td></tr><tr><td align="left" valign="bottom">FtsQ</td><td align="char" char="." valign="bottom">1.71 <xref ref-type="table-fn" rid="table2fn4"><sup>§</sup></xref><sup>,<xref ref-type="table-fn" rid="table2fn2">†</xref></sup></td><td align="char" char="." valign="bottom">1.70 <xref ref-type="table-fn" rid="table2fn9"><sup>§ §</sup></xref></td></tr></tbody></table><table-wrap-foot><fn id="table2fn1"><label>*</label><p>Based on leave-one-out training (using 44 IDPs).</p></fn><fn id="table2fn2"><label>†</label><p>Helix boost applied.</p></fn><fn id="table2fn3"><label>‡</label><p>Based on training by the full training set (45 IDPs).</p></fn><fn id="table2fn4"><label>§</label><p>From <xref ref-type="bibr" rid="bib19">Dey et al., 2022</xref>.</p></fn><fn id="table2fn5"><label>¶</label><p>RMSE is scaled down by a factor of 2.39, to correct for the effect of temperature (MD at 288 K; see <xref ref-type="fig" rid="fig3s1">Figure 3—figure supplement 1C</xref>).</p></fn><fn id="table2fn6"><label>**</label><p>From <xref ref-type="bibr" rid="bib80">Yu and Brüschweiler, 2022</xref>.</p></fn><fn id="table2fn7"><label>††</label><p>RMSE is scaled down by a factor of 2.99, to correct for the effects of temperature and magnetic field (MD at 274 K and 850 MHz; see <xref ref-type="fig" rid="fig3s1">Figure 3—figure supplement 1B</xref>).</p></fn><fn id="table2fn8"><label>‡ ‡</label><p>Originally calculated in <xref ref-type="bibr" rid="bib25">Hicks et al., 2020</xref> with correction in <xref ref-type="bibr" rid="bib26">Hicks et al., 2021</xref>.</p></fn><fn id="table2fn9"><label>§ §</label><p>From <xref ref-type="bibr" rid="bib65">Smrt et al., 2023</xref>.</p></fn></table-wrap-foot></table-wrap><p>Deep-learning models have become very powerful, but they usually have millions of parameters and require millions of protein sequences for training (<xref ref-type="bibr" rid="bib56">Rives et al., 2021</xref>). In contrast, SeqDYN employs a mathematical model with dozens of parameters and requires only dozens of proteins for training. Reduced models (by collapsing amino acids into a small number of distinct types) have even been trained on &lt;10 IDPs to predict propensities for binding nanoparticles (<xref ref-type="bibr" rid="bib35">Li et al., 2020</xref>) or membranes (<xref ref-type="bibr" rid="bib52">Qin et al., 2022</xref>). The mathematical model-based approach may be useful in other applications where data, similar to <inline-formula><mml:math id="inf271"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula>, are limited, including predictions of IDP secondary chemical shifts or residues that bind drug molecules (<xref ref-type="bibr" rid="bib57">Robustelli et al., 2022</xref>) or protein targets, or even in protein design, for example for recognizing an antigenic site or a specific DNA site.</p></sec><sec id="s4" sec-type="methods"><title>Methods</title><sec id="s4-1"><title>Collection of IDPs with measured <inline-formula><mml:math id="inf272"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula></title><p>Starting from six nonhomologous IDPs in our previous MD study (<xref ref-type="bibr" rid="bib19">Dey et al., 2022</xref>), we obtained <inline-formula><mml:math id="inf273"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> data for eight IDPs from the Bimolecular Magnetic Resonance Data Bank (BMRB; <ext-link ext-link-type="uri" xlink:href="https://bmrb.io">https://bmrb.io</ext-link>); data for two other IDPs were from our collaborators (<xref ref-type="bibr" rid="bib25">Hicks et al., 2020</xref>; <xref ref-type="bibr" rid="bib65">Smrt et al., 2023</xref>). Most of the 54 IDPs studied here were from searching the literature. Disorder was judged by dispersion in backbone amide proton chemical shifts, NOE, and SCS. <inline-formula><mml:math id="inf274"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> data that were not available from the authors or BMRB were obtained by digitizing <inline-formula><mml:math id="inf275"><mml:msub><mml:mrow><mml:mi>R</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:math></inline-formula> plots presented in figures of published papers, using WebPlotDigitizer (<ext-link ext-link-type="uri" xlink:href="https://automeris.io/WebPlotDigitizer">https://automeris.io/WebPlotDigitizer</ext-link>; <xref ref-type="bibr" rid="bib58">Rohatgi, 2022</xref>) and further inspected visually.</p><p>Homology of IDPs was checked by sequence alignment using Clustal W (<ext-link ext-link-type="uri" xlink:href="http://www.clustal.org/clustal2">http://www.clustal.org/clustal2</ext-link>; <xref ref-type="bibr" rid="bib33">Larkin et al., 2007</xref>), and presented as a clock-like tree using the ‘ape’ package (<ext-link ext-link-type="uri" xlink:href="http://ape-package.ird.fr">http://ape-package.ird.fr</ext-link>; <xref ref-type="bibr" rid="bib49">Paradis et al., 2004</xref>). IDPs that had discernible homology with the selected training set were removed. Removed IDPs included HOX-SCR and β-synuclein from our previous MD study (<xref ref-type="bibr" rid="bib19">Dey et al., 2022</xref>), due to homology with HOX-DFD and α-synuclein, respectively.</p></sec><sec id="s4-2"><title>Coding for SeqDYN</title><p>The training of SeqDYN was coded in python, similar to our previous work for predicting residue-specific membrane association propensities (ReSMAP; <ext-link ext-link-type="uri" xlink:href="https://zhougroup-uic.github.io/ReSMAPidp/">https://zhougroup-uic.github.io/ReSMAPidp/</ext-link>; <xref ref-type="bibr" rid="bib52">Qin et al., 2022</xref>). The cost function was the sum of mean-squared-errors for the IDPs in the training set. We used the least_squares function in scipy.optimize, with Trust Region Reflective as the minimization algorithm and all parameters restricted to the positive range. For the web server (<ext-link ext-link-type="uri" xlink:href="https://zhougroup-uic.github.io/SeqDYNidp/">https://zhougroup-uic.github.io/SeqDYNidp/</ext-link>; <xref ref-type="bibr" rid="bib53">Qin and Zhou, 2024</xref>), we rewrote the prediction code javascript.</p></sec></sec></body><back><sec sec-type="additional-information" id="s5"><title>Additional information</title><fn-group content-type="competing-interest"><title>Competing interests</title><fn fn-type="COI-statement" id="conf1"><p>No competing interests declared</p></fn></fn-group><fn-group content-type="author-contribution"><title>Author contributions</title><fn fn-type="con" id="con1"><p>Resources, Data curation, Software, Formal analysis, Validation, Investigation, Visualization, Methodology</p></fn><fn fn-type="con" id="con2"><p>Conceptualization, Resources, Data curation, Formal analysis, Supervision, Funding acquisition, Validation, Investigation, Visualization, Methodology, Writing - original draft, Project administration, Writing - review and editing</p></fn></fn-group></sec><sec sec-type="supplementary-material" id="s6"><title>Additional files</title><supplementary-material id="mdar"><label>MDAR checklist</label><media xlink:href="elife-88958-mdarchecklist1-v1.pdf" mimetype="application" mime-subtype="pdf"/></supplementary-material></sec><sec sec-type="data-availability" id="s7"><title>Data availability</title><p>All data generated or analyzed during this study are included in the manuscript and supplementary files; source data have been provided for <xref ref-type="fig" rid="fig3">Figures 3</xref>—<xref ref-type="fig" rid="fig9">9</xref>, <xref ref-type="fig" rid="fig3s1">Figure 3—figure supplement 1</xref>, <xref ref-type="fig" rid="fig4s1">Figure 4—figure supplement 1</xref>, and <xref ref-type="fig" rid="fig5s1">Figure 5—figure supplement 1</xref>.</p></sec><ack id="ack"><title>Acknowledgements</title><p>This work was supported by Grant GM118091 from the National Institutes of Health.</p></ack><ref-list><title>References</title><ref id="bib1"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Abyzov</surname><given-names>A</given-names></name><name><surname>Salvi</surname><given-names>N</given-names></name><name><surname>Schneider</surname><given-names>R</given-names></name><name><surname>Maurin</surname><given-names>D</given-names></name><name><surname>Ruigrok</surname><given-names>RWH</given-names></name><name><surname>Jensen</surname><given-names>MR</given-names></name><name><surname>Blackledge</surname><given-names>M</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Identification of dynamic modes in an intrinsically disordered protein using temperature-dependent NMR relaxation</article-title><source>Journal of the American Chemical Society</source><volume>138</volume><fpage>6240</fpage><lpage>6251</lpage><pub-id pub-id-type="doi">10.1021/jacs.6b02424</pub-id><pub-id pub-id-type="pmid">27112095</pub-id></element-citation></ref><ref id="bib2"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ahmed</surname><given-names>MAM</given-names></name><name><surname>Bamm</surname><given-names>VV</given-names></name><name><surname>Harauz</surname><given-names>G</given-names></name><name><surname>Ladizhansky</surname><given-names>V</given-names></name></person-group><year iso-8601-date="2007">2007</year><article-title>The BG21 isoform of Golli myelin basic protein is intrinsically disordered with a highly flexible amino-terminal domain</article-title><source>Biochemistry</source><volume>46</volume><fpage>9700</fpage><lpage>9712</lpage><pub-id pub-id-type="doi">10.1021/bi700632x</pub-id><pub-id pub-id-type="pmid">17676872</pub-id></element-citation></ref><ref id="bib3"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Bafaro</surname><given-names>EM</given-names></name><name><surname>Maciejewski</surname><given-names>MW</given-names></name><name><surname>Hoch</surname><given-names>JC</given-names></name><name><surname>Dempski</surname><given-names>RE</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Concomitant disorder and high-affinity zinc binding in the human zinc- and iron-regulated transport protein 4 intracellular loop</article-title><source>Protein Science</source><volume>28</volume><fpage>868</fpage><lpage>880</lpage><pub-id pub-id-type="doi">10.1002/pro.3591</pub-id><pub-id pub-id-type="pmid">30793391</pub-id></element-citation></ref><ref id="bib4"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ban</surname><given-names>D</given-names></name><name><surname>Iconaru</surname><given-names>LI</given-names></name><name><surname>Ramanathan</surname><given-names>A</given-names></name><name><surname>Zuo</surname><given-names>J</given-names></name><name><surname>Kriwacki</surname><given-names>RW</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>A small molecule causes a population shift in the conformational landscape of an intrinsically disordered protein</article-title><source>Journal of the American Chemical Society</source><volume>139</volume><fpage>13692</fpage><lpage>13700</lpage><pub-id pub-id-type="doi">10.1021/jacs.7b01380</pub-id><pub-id pub-id-type="pmid">28885015</pub-id></element-citation></ref><ref id="bib5"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Barré</surname><given-names>P</given-names></name><name><surname>Eliezer</surname><given-names>D</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>Structural transitions in tau k18 on micelle binding suggest a hierarchy in the efficacy of individual microtubule-binding repeats in filament nucleation</article-title><source>Protein Science</source><volume>22</volume><fpage>1037</fpage><lpage>1048</lpage><pub-id pub-id-type="doi">10.1002/pro.2290</pub-id><pub-id pub-id-type="pmid">23740819</pub-id></element-citation></ref><ref id="bib6"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Bernadó</surname><given-names>P</given-names></name><name><surname>Blackledge</surname><given-names>M</given-names></name></person-group><year iso-8601-date="2009">2009</year><article-title>A self-consistent description of the conformational behavior of chemically denatured proteins from NMR and small angle scattering</article-title><source>Biophysical Journal</source><volume>97</volume><fpage>2839</fpage><lpage>2845</lpage><pub-id pub-id-type="doi">10.1016/j.bpj.2009.08.044</pub-id><pub-id pub-id-type="pmid">19917239</pub-id></element-citation></ref><ref id="bib7"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Borgia</surname><given-names>A</given-names></name><name><surname>Borgia</surname><given-names>MB</given-names></name><name><surname>Bugge</surname><given-names>K</given-names></name><name><surname>Kissling</surname><given-names>VM</given-names></name><name><surname>Heidarsson</surname><given-names>PO</given-names></name><name><surname>Fernandes</surname><given-names>CB</given-names></name><name><surname>Sottini</surname><given-names>A</given-names></name><name><surname>Soranno</surname><given-names>A</given-names></name><name><surname>Buholzer</surname><given-names>KJ</given-names></name><name><surname>Nettels</surname><given-names>D</given-names></name><name><surname>Kragelund</surname><given-names>BB</given-names></name><name><surname>Best</surname><given-names>RB</given-names></name><name><surname>Schuler</surname><given-names>B</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Extreme disorder in an ultrahigh-affinity protein complex</article-title><source>Nature</source><volume>555</volume><fpage>61</fpage><lpage>66</lpage><pub-id pub-id-type="doi">10.1038/nature25762</pub-id><pub-id pub-id-type="pmid">29466338</pub-id></element-citation></ref><ref id="bib8"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Burke</surname><given-names>KA</given-names></name><name><surname>Janke</surname><given-names>AM</given-names></name><name><surname>Rhine</surname><given-names>CL</given-names></name><name><surname>Fawzi</surname><given-names>NL</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>Residue-by-residue view of in vitro FUS granules that bind the C-terminal domain of RNA polymerase II</article-title><source>Molecular Cell</source><volume>60</volume><fpage>231</fpage><lpage>241</lpage><pub-id pub-id-type="doi">10.1016/j.molcel.2015.09.006</pub-id><pub-id pub-id-type="pmid">26455390</pub-id></element-citation></ref><ref id="bib9"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Camacho-Zarco</surname><given-names>AR</given-names></name><name><surname>Schnapka</surname><given-names>V</given-names></name><name><surname>Guseva</surname><given-names>S</given-names></name><name><surname>Abyzov</surname><given-names>A</given-names></name><name><surname>Adamski</surname><given-names>W</given-names></name><name><surname>Milles</surname><given-names>S</given-names></name><name><surname>Jensen</surname><given-names>MR</given-names></name><name><surname>Zidek</surname><given-names>L</given-names></name><name><surname>Salvi</surname><given-names>N</given-names></name><name><surname>Blackledge</surname><given-names>M</given-names></name></person-group><year iso-8601-date="2022">2022</year><article-title>NMR provides unique insight into the functional dynamics and interactions of intrinsically disordered proteins</article-title><source>Chemical Reviews</source><volume>122</volume><fpage>9331</fpage><lpage>9356</lpage><pub-id pub-id-type="doi">10.1021/acs.chemrev.1c01023</pub-id><pub-id pub-id-type="pmid">35446534</pub-id></element-citation></ref><ref id="bib10"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Chandrashekaran</surname><given-names>IR</given-names></name><name><surname>Mohanty</surname><given-names>B</given-names></name><name><surname>Linossi</surname><given-names>EM</given-names></name><name><surname>Dagley</surname><given-names>LF</given-names></name><name><surname>Leung</surname><given-names>EWW</given-names></name><name><surname>Murphy</surname><given-names>JM</given-names></name><name><surname>Babon</surname><given-names>JJ</given-names></name><name><surname>Nicholson</surname><given-names>SE</given-names></name><name><surname>Norton</surname><given-names>RS</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>Structure and functional characterization of the conserved jak interaction region in the intrinsically disordered N-terminus of SOCS5</article-title><source>Biochemistry</source><volume>54</volume><fpage>4672</fpage><lpage>4682</lpage><pub-id pub-id-type="doi">10.1021/acs.biochem.5b00619</pub-id><pub-id pub-id-type="pmid">26173083</pub-id></element-citation></ref><ref id="bib11"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Cho</surname><given-names>MK</given-names></name><name><surname>Kim</surname><given-names>HY</given-names></name><name><surname>Bernado</surname><given-names>P</given-names></name><name><surname>Fernandez</surname><given-names>CO</given-names></name><name><surname>Blackledge</surname><given-names>M</given-names></name><name><surname>Zweckstetter</surname><given-names>M</given-names></name></person-group><year iso-8601-date="2007">2007</year><article-title>Amino acid bulkiness defines the local conformations and dynamics of natively unfolded alpha-synuclein and tau</article-title><source>Journal of the American Chemical Society</source><volume>129</volume><fpage>3032</fpage><lpage>3033</lpage><pub-id pub-id-type="doi">10.1021/ja067482k</pub-id><pub-id pub-id-type="pmid">17315997</pub-id></element-citation></ref><ref id="bib12"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Cho</surname><given-names>HY</given-names></name><name><surname>Ul Mushtaq</surname><given-names>A</given-names></name><name><surname>Lee</surname><given-names>JY</given-names></name><name><surname>Kim</surname><given-names>DG</given-names></name><name><surname>Seok</surname><given-names>MS</given-names></name><name><surname>Jang</surname><given-names>M</given-names></name><name><surname>Han</surname><given-names>B-W</given-names></name><name><surname>Kim</surname><given-names>S</given-names></name><name><surname>Jeon</surname><given-names>YH</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>Characterization of the interaction between lysyl-tRNA synthetase and laminin receptor by NMR</article-title><source>FEBS Letters</source><volume>588</volume><fpage>2851</fpage><lpage>2858</lpage><pub-id pub-id-type="doi">10.1016/j.febslet.2014.06.048</pub-id><pub-id pub-id-type="pmid">24983501</pub-id></element-citation></ref><ref id="bib13"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Cino</surname><given-names>EA</given-names></name><name><surname>Karttunen</surname><given-names>M</given-names></name><name><surname>Choy</surname><given-names>WY</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>Effects of molecular crowding on the dynamics of intrinsically disordered proteins</article-title><source>PLOS ONE</source><volume>7</volume><elocation-id>e49876</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pone.0049876</pub-id><pub-id pub-id-type="pmid">23189168</pub-id></element-citation></ref><ref id="bib14"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Conicella</surname><given-names>AE</given-names></name><name><surname>Zerze</surname><given-names>GH</given-names></name><name><surname>Mittal</surname><given-names>J</given-names></name><name><surname>Fawzi</surname><given-names>NL</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>ALS mutations disrupt phase separation mediated by α-Helical Structure in the TDP-43 Low-Complexity C-terminal domain</article-title><source>Structure</source><volume>24</volume><fpage>1537</fpage><lpage>1549</lpage><pub-id pub-id-type="doi">10.1016/j.str.2016.07.007</pub-id><pub-id pub-id-type="pmid">27545621</pub-id></element-citation></ref><ref id="bib15"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Cook</surname><given-names>EC</given-names></name><name><surname>Sahu</surname><given-names>D</given-names></name><name><surname>Bastidas</surname><given-names>M</given-names></name><name><surname>Showalter</surname><given-names>SA</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Solution ensemble of the C-terminal domain from the transcription factor Pdx1 resembles an excluded volume polymer</article-title><source>The Journal of Physical Chemistry. B</source><volume>123</volume><fpage>106</fpage><lpage>116</lpage><pub-id pub-id-type="doi">10.1021/acs.jpcb.8b10051</pub-id><pub-id pub-id-type="pmid">30525611</pub-id></element-citation></ref><ref id="bib16"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Csizmok</surname><given-names>V</given-names></name><name><surname>Felli</surname><given-names>IC</given-names></name><name><surname>Tompa</surname><given-names>P</given-names></name><name><surname>Banci</surname><given-names>L</given-names></name><name><surname>Bertini</surname><given-names>I</given-names></name></person-group><year iso-8601-date="2008">2008</year><article-title>Structural and dynamic characterization of intrinsically disordered human securin by NMR spectroscopy</article-title><source>Journal of the American Chemical Society</source><volume>130</volume><fpage>16873</fpage><lpage>16879</lpage><pub-id pub-id-type="doi">10.1021/ja805510b</pub-id><pub-id pub-id-type="pmid">19053469</pub-id></element-citation></ref><ref id="bib17"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>De Avila</surname><given-names>M</given-names></name><name><surname>Vassall</surname><given-names>KA</given-names></name><name><surname>Smith</surname><given-names>GST</given-names></name><name><surname>Bamm</surname><given-names>VV</given-names></name><name><surname>Harauz</surname><given-names>G</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>The proline-rich region of 18.5 kDa myelin basic protein binds to the SH3-domain of Fyn tyrosine kinase with the aid of an upstream segment to form a dynamic complex in vitro</article-title><source>Bioscience Reports</source><volume>34</volume><elocation-id>e00157</elocation-id><pub-id pub-id-type="doi">10.1042/BSR20140149</pub-id><pub-id pub-id-type="pmid">25343306</pub-id></element-citation></ref><ref id="bib18"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Delaforge</surname><given-names>E</given-names></name><name><surname>Kragelj</surname><given-names>J</given-names></name><name><surname>Tengo</surname><given-names>L</given-names></name><name><surname>Palencia</surname><given-names>A</given-names></name><name><surname>Milles</surname><given-names>S</given-names></name><name><surname>Bouvignies</surname><given-names>G</given-names></name><name><surname>Salvi</surname><given-names>N</given-names></name><name><surname>Blackledge</surname><given-names>M</given-names></name><name><surname>Jensen</surname><given-names>MR</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Deciphering the dynamic interaction profile of an intrinsically disordered protein by nmr exchange spectroscopy</article-title><source>Journal of the American Chemical Society</source><volume>140</volume><fpage>1148</fpage><lpage>1158</lpage><pub-id pub-id-type="doi">10.1021/jacs.7b12407</pub-id><pub-id pub-id-type="pmid">29276882</pub-id></element-citation></ref><ref id="bib19"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Dey</surname><given-names>S</given-names></name><name><surname>MacAinsh</surname><given-names>M</given-names></name><name><surname>Zhou</surname><given-names>HX</given-names></name></person-group><year iso-8601-date="2022">2022</year><article-title>Sequence-dependent backbone dynamics of intrinsically disordered proteins</article-title><source>Journal of Chemical Theory and Computation</source><volume>18</volume><fpage>6310</fpage><lpage>6323</lpage><pub-id pub-id-type="doi">10.1021/acs.jctc.2c00328</pub-id><pub-id pub-id-type="pmid">36084347</pub-id></element-citation></ref><ref id="bib20"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ebert</surname><given-names>MO</given-names></name><name><surname>Bae</surname><given-names>SH</given-names></name><name><surname>Dyson</surname><given-names>HJ</given-names></name><name><surname>Wright</surname><given-names>PE</given-names></name></person-group><year iso-8601-date="2008">2008</year><article-title>NMR relaxation study of the complex formed between CBP and the activation domain of the nuclear hormone receptor coactivator ACTR</article-title><source>Biochemistry</source><volume>47</volume><fpage>1299</fpage><lpage>1308</lpage><pub-id pub-id-type="doi">10.1021/bi701767j</pub-id><pub-id pub-id-type="pmid">18177052</pub-id></element-citation></ref><ref id="bib21"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Feichtinger</surname><given-names>M</given-names></name><name><surname>Beier</surname><given-names>A</given-names></name><name><surname>Migotti</surname><given-names>M</given-names></name><name><surname>Schmid</surname><given-names>M</given-names></name><name><surname>Bokhovchuk</surname><given-names>F</given-names></name><name><surname>Chène</surname><given-names>P</given-names></name><name><surname>Konrat</surname><given-names>R</given-names></name></person-group><year iso-8601-date="2022">2022</year><article-title>Long-range structural preformation in yes-associated protein precedes encounter complex formation with TEAD</article-title><source>iScience</source><volume>25</volume><elocation-id>104099</elocation-id><pub-id pub-id-type="doi">10.1016/j.isci.2022.104099</pub-id></element-citation></ref><ref id="bib22"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Feldman</surname><given-names>HJ</given-names></name><name><surname>Hogue</surname><given-names>CWV</given-names></name></person-group><year iso-8601-date="2002">2002</year><article-title>Probabilistic sampling of protein conformations: new hope for brute force?</article-title><source>Proteins</source><volume>46</volume><fpage>8</fpage><lpage>23</lpage><pub-id pub-id-type="pmid">11746699</pub-id></element-citation></ref><ref id="bib23"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Gruber</surname><given-names>T</given-names></name><name><surname>Lewitzky</surname><given-names>M</given-names></name><name><surname>Machner</surname><given-names>L</given-names></name><name><surname>Weininger</surname><given-names>U</given-names></name><name><surname>Feller</surname><given-names>SM</given-names></name><name><surname>Balbach</surname><given-names>J</given-names></name></person-group><year iso-8601-date="2022">2022</year><article-title>Macromolecular crowding induces a binding competent transient structure in intrinsically disordered Gab1</article-title><source>Journal of Molecular Biology</source><volume>434</volume><elocation-id>167407</elocation-id><pub-id pub-id-type="doi">10.1016/j.jmb.2021.167407</pub-id><pub-id pub-id-type="pmid">34929201</pub-id></element-citation></ref><ref id="bib24"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Harris</surname><given-names>J</given-names></name><name><surname>Shadrina</surname><given-names>M</given-names></name><name><surname>Oliver</surname><given-names>C</given-names></name><name><surname>Vogel</surname><given-names>J</given-names></name><name><surname>Mittermaier</surname><given-names>A</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Concerted millisecond timescale dynamics in the intrinsically disordered carboxyl terminus of γ-tubulin induced by mutation of a conserved tyrosine residue</article-title><source>Protein Science</source><volume>27</volume><fpage>531</fpage><lpage>545</lpage><pub-id pub-id-type="doi">10.1002/pro.3345</pub-id><pub-id pub-id-type="pmid">29127738</pub-id></element-citation></ref><ref id="bib25"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hicks</surname><given-names>A</given-names></name><name><surname>Escobar</surname><given-names>CA</given-names></name><name><surname>Cross</surname><given-names>TA</given-names></name><name><surname>Zhou</surname><given-names>HX</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>Sequence-dependent correlated segments in the intrinsically disordered region of ChiZ</article-title><source>Biomolecules</source><volume>10</volume><fpage>1</fpage><lpage>23</lpage><pub-id pub-id-type="doi">10.3390/biom10060946</pub-id><pub-id pub-id-type="pmid">32585849</pub-id></element-citation></ref><ref id="bib26"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hicks</surname><given-names>A</given-names></name><name><surname>MacAinsh</surname><given-names>M</given-names></name><name><surname>Zhou</surname><given-names>HX</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>Removing thermostat distortions of protein dynamics in constant-temperature molecular dynamics simulations</article-title><source>Journal of Chemical Theory and Computation</source><volume>17</volume><fpage>5920</fpage><lpage>5932</lpage><pub-id pub-id-type="doi">10.1021/acs.jctc.1c00448</pub-id><pub-id pub-id-type="pmid">34464112</pub-id></element-citation></ref><ref id="bib27"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Janke</surname><given-names>AM</given-names></name><name><surname>Seo</surname><given-names>DH</given-names></name><name><surname>Rahmanian</surname><given-names>V</given-names></name><name><surname>Conicella</surname><given-names>AE</given-names></name><name><surname>Mathews</surname><given-names>KL</given-names></name><name><surname>Burke</surname><given-names>KA</given-names></name><name><surname>Mittal</surname><given-names>J</given-names></name><name><surname>Fawzi</surname><given-names>NL</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Lysines in the RNA polymerase II C-terminal domain contribute to TAF15 fibril recruitment</article-title><source>Biochemistry</source><volume>57</volume><fpage>2549</fpage><lpage>2563</lpage><pub-id pub-id-type="doi">10.1021/acs.biochem.7b00310</pub-id><pub-id pub-id-type="pmid">28945358</pub-id></element-citation></ref><ref id="bib28"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Jenner</surname><given-names>M</given-names></name><name><surname>Kosol</surname><given-names>S</given-names></name><name><surname>Griffiths</surname><given-names>D</given-names></name><name><surname>Prasongpholchai</surname><given-names>P</given-names></name><name><surname>Manzi</surname><given-names>L</given-names></name><name><surname>Barrow</surname><given-names>AS</given-names></name><name><surname>Moses</surname><given-names>JE</given-names></name><name><surname>Oldham</surname><given-names>NJ</given-names></name><name><surname>Lewandowski</surname><given-names>JR</given-names></name><name><surname>Challis</surname><given-names>GL</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Mechanism of intersubunit ketosynthase-dehydratase interaction in polyketide synthases</article-title><source>Nature Chemical Biology</source><volume>14</volume><fpage>270</fpage><lpage>275</lpage><pub-id pub-id-type="doi">10.1038/nchembio.2549</pub-id><pub-id pub-id-type="pmid">29309054</pub-id></element-citation></ref><ref id="bib29"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Jensen</surname><given-names>MR</given-names></name><name><surname>Houben</surname><given-names>K</given-names></name><name><surname>Lescop</surname><given-names>E</given-names></name><name><surname>Blanchard</surname><given-names>L</given-names></name><name><surname>Ruigrok</surname><given-names>RWH</given-names></name><name><surname>Blackledge</surname><given-names>M</given-names></name></person-group><year iso-8601-date="2008">2008</year><article-title>Quantitative conformational analysis of partially folded proteins from residual dipolar couplings: application to the molecular recognition element of Sendai virus nucleoprotein</article-title><source>Journal of the American Chemical Society</source><volume>130</volume><fpage>8055</fpage><lpage>8061</lpage><pub-id pub-id-type="doi">10.1021/ja801332d</pub-id><pub-id pub-id-type="pmid">18507376</pub-id></element-citation></ref><ref id="bib30"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kiss</surname><given-names>R</given-names></name><name><surname>Kovács</surname><given-names>D</given-names></name><name><surname>Tompa</surname><given-names>P</given-names></name><name><surname>Perczel</surname><given-names>A</given-names></name></person-group><year iso-8601-date="2008">2008</year><article-title>Local structural preferences of calpastatin, the intrinsically unstructured protein inhibitor of calpain</article-title><source>Biochemistry</source><volume>47</volume><fpage>6936</fpage><lpage>6945</lpage><pub-id pub-id-type="doi">10.1021/bi800201a</pub-id><pub-id pub-id-type="pmid">18537264</pub-id></element-citation></ref><ref id="bib31"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Klein-Seetharaman</surname><given-names>J</given-names></name><name><surname>Oikawa</surname><given-names>M</given-names></name><name><surname>Grimshaw</surname><given-names>SB</given-names></name><name><surname>Wirmer</surname><given-names>J</given-names></name><name><surname>Duchardt</surname><given-names>E</given-names></name><name><surname>Ueda</surname><given-names>T</given-names></name><name><surname>Imoto</surname><given-names>T</given-names></name><name><surname>Smith</surname><given-names>LJ</given-names></name><name><surname>Dobson</surname><given-names>CM</given-names></name><name><surname>Schwalbe</surname><given-names>H</given-names></name></person-group><year iso-8601-date="2002">2002</year><article-title>Long-range interactions within a nonnative protein</article-title><source>Science</source><volume>295</volume><fpage>1719</fpage><lpage>1722</lpage><pub-id pub-id-type="doi">10.1126/science.1067680</pub-id><pub-id pub-id-type="pmid">11872841</pub-id></element-citation></ref><ref id="bib32"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Lakomek</surname><given-names>NA</given-names></name><name><surname>Yavuz</surname><given-names>H</given-names></name><name><surname>Jahn</surname><given-names>R</given-names></name><name><surname>Pérez-Lara</surname><given-names>Á</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Structural dynamics and transient lipid binding of synaptobrevin-2 tune SNARE assembly and membrane fusion</article-title><source>PNAS</source><volume>116</volume><fpage>8699</fpage><lpage>8708</lpage><pub-id pub-id-type="doi">10.1073/pnas.1813194116</pub-id><pub-id pub-id-type="pmid">30975750</pub-id></element-citation></ref><ref id="bib33"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Larkin</surname><given-names>MA</given-names></name><name><surname>Blackshields</surname><given-names>G</given-names></name><name><surname>Brown</surname><given-names>NP</given-names></name><name><surname>Chenna</surname><given-names>R</given-names></name><name><surname>McGettigan</surname><given-names>PA</given-names></name><name><surname>McWilliam</surname><given-names>H</given-names></name><name><surname>Valentin</surname><given-names>F</given-names></name><name><surname>Wallace</surname><given-names>IM</given-names></name><name><surname>Wilm</surname><given-names>A</given-names></name><name><surname>Lopez</surname><given-names>R</given-names></name><name><surname>Thompson</surname><given-names>JD</given-names></name><name><surname>Gibson</surname><given-names>TJ</given-names></name><name><surname>Higgins</surname><given-names>DG</given-names></name></person-group><year iso-8601-date="2007">2007</year><article-title>Clustal W and clustal X version 2.0</article-title><source>Bioinformatics</source><volume>23</volume><fpage>2947</fpage><lpage>2948</lpage><pub-id pub-id-type="doi">10.1093/bioinformatics/btm404</pub-id><pub-id pub-id-type="pmid">17846036</pub-id></element-citation></ref><ref id="bib34"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Lawrence</surname><given-names>CW</given-names></name><name><surname>Showalter</surname><given-names>SA</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>Carbon-detected (15)N NMR spin relaxation of an intrinsically disordered protein: FCP1 dynamics unbound and in complex with RAP74</article-title><source>The Journal of Physical Chemistry Letters</source><volume>3</volume><fpage>1409</fpage><lpage>1413</lpage><pub-id pub-id-type="doi">10.1021/jz300432e</pub-id><pub-id pub-id-type="pmid">26286791</pub-id></element-citation></ref><ref id="bib35"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Li</surname><given-names>D-W</given-names></name><name><surname>Xie</surname><given-names>M</given-names></name><name><surname>Brüschweiler</surname><given-names>R</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>Quantitative cooperative binding model for intrinsically disordered proteins interacting with nanomaterials</article-title><source>Journal of the American Chemical Society</source><volume>142</volume><fpage>10730</fpage><lpage>10738</lpage><pub-id pub-id-type="doi">10.1021/jacs.0c01885</pub-id><pub-id pub-id-type="pmid">32426975</pub-id></element-citation></ref><ref id="bib36"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Lopes</surname><given-names>FC</given-names></name><name><surname>Dobrovolska</surname><given-names>O</given-names></name><name><surname>Real-Guerra</surname><given-names>R</given-names></name><name><surname>Broll</surname><given-names>V</given-names></name><name><surname>Zambelli</surname><given-names>B</given-names></name><name><surname>Musiani</surname><given-names>F</given-names></name><name><surname>Uversky</surname><given-names>VN</given-names></name><name><surname>Carlini</surname><given-names>CR</given-names></name><name><surname>Ciurli</surname><given-names>S</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>Pliable natural biocide: Jaburetox is an intrinsically disordered insecticidal and fungicidal polypeptide derived from jack bean urease</article-title><source>The FEBS Journal</source><volume>282</volume><fpage>1043</fpage><lpage>1064</lpage><pub-id pub-id-type="doi">10.1111/febs.13201</pub-id><pub-id pub-id-type="pmid">25605001</pub-id></element-citation></ref><ref id="bib37"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Maiti</surname><given-names>S</given-names></name><name><surname>Acharya</surname><given-names>B</given-names></name><name><surname>Boorla</surname><given-names>VS</given-names></name><name><surname>Manna</surname><given-names>B</given-names></name><name><surname>Ghosh</surname><given-names>A</given-names></name><name><surname>De</surname><given-names>S</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Dynamic studies on intrinsically disordered regions of two paralogous transcription factors reveal rigid segments with important biological functions</article-title><source>Journal of Molecular Biology</source><volume>431</volume><fpage>1353</fpage><lpage>1369</lpage><pub-id pub-id-type="doi">10.1016/j.jmb.2019.02.021</pub-id><pub-id pub-id-type="pmid">30802457</pub-id></element-citation></ref><ref id="bib38"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Malki</surname><given-names>A</given-names></name><name><surname>Teulon</surname><given-names>J-M</given-names></name><name><surname>Camacho-Zarco</surname><given-names>AR</given-names></name><name><surname>Chen</surname><given-names>S-WW</given-names></name><name><surname>Adamski</surname><given-names>W</given-names></name><name><surname>Maurin</surname><given-names>D</given-names></name><name><surname>Salvi</surname><given-names>N</given-names></name><name><surname>Pellequer</surname><given-names>J-L</given-names></name><name><surname>Blackledge</surname><given-names>M</given-names></name></person-group><year iso-8601-date="2022">2022</year><article-title>Intrinsically disordered tardigrade proteins self-assemble into fibrous gels in response to environmental stress</article-title><source>Angewandte Chemie</source><volume>61</volume><elocation-id>e202109961</elocation-id><pub-id pub-id-type="doi">10.1002/anie.202109961</pub-id><pub-id pub-id-type="pmid">34750927</pub-id></element-citation></ref><ref id="bib39"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Martin</surname><given-names>EW</given-names></name><name><surname>Holehouse</surname><given-names>AS</given-names></name><name><surname>Grace</surname><given-names>CR</given-names></name><name><surname>Hughes</surname><given-names>A</given-names></name><name><surname>Pappu</surname><given-names>RV</given-names></name><name><surname>Mittag</surname><given-names>T</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Sequence determinants of the conformational properties of an intrinsically disordered protein prior to and upon multisite phosphorylation</article-title><source>Journal of the American Chemical Society</source><volume>138</volume><fpage>15323</fpage><lpage>15335</lpage><pub-id pub-id-type="doi">10.1021/jacs.6b10272</pub-id><pub-id pub-id-type="pmid">27807972</pub-id></element-citation></ref><ref id="bib40"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Martin</surname><given-names>EW</given-names></name><name><surname>Holehouse</surname><given-names>AS</given-names></name><name><surname>Peran</surname><given-names>I</given-names></name><name><surname>Farag</surname><given-names>M</given-names></name><name><surname>Incicco</surname><given-names>JJ</given-names></name><name><surname>Bremer</surname><given-names>A</given-names></name><name><surname>Grace</surname><given-names>CR</given-names></name><name><surname>Soranno</surname><given-names>A</given-names></name><name><surname>Pappu</surname><given-names>RV</given-names></name><name><surname>Mittag</surname><given-names>T</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>Valence and patterning of aromatic residues determine the phase behavior of prion-like domains</article-title><source>Science</source><volume>367</volume><fpage>694</fpage><lpage>699</lpage><pub-id pub-id-type="doi">10.1126/science.aaw8653</pub-id><pub-id pub-id-type="pmid">32029630</pub-id></element-citation></ref><ref id="bib41"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Mateos</surname><given-names>B</given-names></name><name><surname>Conrad-Billroth</surname><given-names>C</given-names></name><name><surname>Schiavina</surname><given-names>M</given-names></name><name><surname>Beier</surname><given-names>A</given-names></name><name><surname>Kontaxis</surname><given-names>G</given-names></name><name><surname>Konrat</surname><given-names>R</given-names></name><name><surname>Felli</surname><given-names>IC</given-names></name><name><surname>Pierattelli</surname><given-names>R</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>The ambivalent role of proline residues in an intrinsically disordered protein: from disorder promoters to compaction facilitators</article-title><source>Journal of Molecular Biology</source><volume>432</volume><fpage>3093</fpage><lpage>3111</lpage><pub-id pub-id-type="doi">10.1016/j.jmb.2019.11.015</pub-id><pub-id pub-id-type="pmid">31794728</pub-id></element-citation></ref><ref id="bib42"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>McGuffin</surname><given-names>LJ</given-names></name><name><surname>Bryson</surname><given-names>K</given-names></name><name><surname>Jones</surname><given-names>DT</given-names></name></person-group><year iso-8601-date="2000">2000</year><article-title>The PSIPRED protein structure prediction server</article-title><source>Bioinformatics</source><volume>16</volume><fpage>404</fpage><lpage>405</lpage><pub-id pub-id-type="doi">10.1093/bioinformatics/16.4.404</pub-id><pub-id pub-id-type="pmid">10869041</pub-id></element-citation></ref><ref id="bib43"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Milles</surname><given-names>S</given-names></name><name><surname>Jensen</surname><given-names>MR</given-names></name><name><surname>Lazert</surname><given-names>C</given-names></name><name><surname>Guseva</surname><given-names>S</given-names></name><name><surname>Ivashchenko</surname><given-names>S</given-names></name><name><surname>Communie</surname><given-names>G</given-names></name><name><surname>Maurin</surname><given-names>D</given-names></name><name><surname>Gerlier</surname><given-names>D</given-names></name><name><surname>Ruigrok</surname><given-names>RWH</given-names></name><name><surname>Blackledge</surname><given-names>M</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>An ultraweak interaction in the intrinsically disordered replication machinery is essential for measles virus function</article-title><source>Science Advances</source><volume>4</volume><elocation-id>eaat7778</elocation-id><pub-id pub-id-type="doi">10.1126/sciadv.aat7778</pub-id><pub-id pub-id-type="pmid">30140745</pub-id></element-citation></ref><ref id="bib44"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Mittag</surname><given-names>T</given-names></name><name><surname>Marsh</surname><given-names>J</given-names></name><name><surname>Grishaev</surname><given-names>A</given-names></name><name><surname>Orlicky</surname><given-names>S</given-names></name><name><surname>Lin</surname><given-names>H</given-names></name><name><surname>Sicheri</surname><given-names>F</given-names></name><name><surname>Tyers</surname><given-names>M</given-names></name><name><surname>Forman-Kay</surname><given-names>JD</given-names></name></person-group><year iso-8601-date="2010">2010</year><article-title>Structure/function implications in a dynamic complex of the intrinsically disordered Sic1 with the Cdc4 subunit of an SCF ubiquitin ligase</article-title><source>Structure</source><volume>18</volume><fpage>494</fpage><lpage>506</lpage><pub-id pub-id-type="doi">10.1016/j.str.2010.01.020</pub-id><pub-id pub-id-type="pmid">20399186</pub-id></element-citation></ref><ref id="bib45"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Mokhtarzada</surname><given-names>S</given-names></name><name><surname>Yu</surname><given-names>C</given-names></name><name><surname>Brickenden</surname><given-names>A</given-names></name><name><surname>Choy</surname><given-names>WY</given-names></name></person-group><year iso-8601-date="2011">2011</year><article-title>Structural characterization of partially disordered human Chibby: insights into its function in the Wnt-signaling pathway</article-title><source>Biochemistry</source><volume>50</volume><fpage>715</fpage><lpage>726</lpage><pub-id pub-id-type="doi">10.1021/bi101236z</pub-id><pub-id pub-id-type="pmid">21182262</pub-id></element-citation></ref><ref id="bib46"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Murrali</surname><given-names>MG</given-names></name><name><surname>Piai</surname><given-names>A</given-names></name><name><surname>Bermel</surname><given-names>W</given-names></name><name><surname>Felli</surname><given-names>IC</given-names></name><name><surname>Pierattelli</surname><given-names>R</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Proline fingerprint in intrinsically disordered proteins</article-title><source>Chembiochem</source><volume>19</volume><fpage>1625</fpage><lpage>1629</lpage><pub-id pub-id-type="doi">10.1002/cbic.201800172</pub-id><pub-id pub-id-type="pmid">29790640</pub-id></element-citation></ref><ref id="bib47"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Neira</surname><given-names>JL</given-names></name><name><surname>Palomino-Schätzlein</surname><given-names>M</given-names></name><name><surname>Ricci</surname><given-names>C</given-names></name><name><surname>Ortore</surname><given-names>MG</given-names></name><name><surname>Rizzuti</surname><given-names>B</given-names></name><name><surname>Iovanna</surname><given-names>JL</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Dynamics of the intrinsically disordered protein NUPR1 in isolation and in its fuzzy complexes with DNA and prothymosin α</article-title><source>Biochimica et Biophysica Acta - Proteins and Proteomics</source><volume>1867</volume><elocation-id>140252</elocation-id><pub-id pub-id-type="doi">10.1016/j.bbapap.2019.07.005</pub-id></element-citation></ref><ref id="bib48"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Olivieri</surname><given-names>C</given-names></name><name><surname>Wang</surname><given-names>Y</given-names></name><name><surname>Li</surname><given-names>GC</given-names></name><name><surname>V S</surname><given-names>M</given-names></name><name><surname>Kim</surname><given-names>J</given-names></name><name><surname>Stultz</surname><given-names>BR</given-names></name><name><surname>Neibergall</surname><given-names>M</given-names></name><name><surname>Porcelli</surname><given-names>F</given-names></name><name><surname>Muretta</surname><given-names>JM</given-names></name><name><surname>Thomas</surname><given-names>DD</given-names></name><name><surname>Gao</surname><given-names>J</given-names></name><name><surname>Blumenthal</surname><given-names>DK</given-names></name><name><surname>Taylor</surname><given-names>SS</given-names></name><name><surname>Veglia</surname><given-names>G</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>Multi-state recognition pathway of the intrinsically disordered protein kinase inhibitor by protein kinase A</article-title><source>eLife</source><volume>9</volume><elocation-id>e55607</elocation-id><pub-id pub-id-type="doi">10.7554/eLife.55607</pub-id><pub-id pub-id-type="pmid">32338601</pub-id></element-citation></ref><ref id="bib49"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Paradis</surname><given-names>E</given-names></name><name><surname>Claude</surname><given-names>J</given-names></name><name><surname>Strimmer</surname><given-names>K</given-names></name></person-group><year iso-8601-date="2004">2004</year><article-title>APE: analyses of phylogenetics and evolution in R language</article-title><source>Bioinformatics</source><volume>20</volume><fpage>289</fpage><lpage>290</lpage><pub-id pub-id-type="doi">10.1093/bioinformatics/btg412</pub-id><pub-id pub-id-type="pmid">14734327</pub-id></element-citation></ref><ref id="bib50"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Patel</surname><given-names>S</given-names></name><name><surname>Ramanujam</surname><given-names>V</given-names></name><name><surname>Srivastava</surname><given-names>AK</given-names></name><name><surname>Chary</surname><given-names>KVR</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>Conformational propensities and dynamics of a βγ-crystallin, an intrinsically disordered protein</article-title><source>Physical Chemistry Chemical Physics</source><volume>16</volume><elocation-id>12703</elocation-id><pub-id pub-id-type="doi">10.1039/c3cp53558d</pub-id></element-citation></ref><ref id="bib51"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Piai</surname><given-names>A</given-names></name><name><surname>Calçada</surname><given-names>EO</given-names></name><name><surname>Tarenzi</surname><given-names>T</given-names></name><name><surname>del Grande</surname><given-names>A</given-names></name><name><surname>Varadi</surname><given-names>M</given-names></name><name><surname>Tompa</surname><given-names>P</given-names></name><name><surname>Felli</surname><given-names>IC</given-names></name><name><surname>Pierattelli</surname><given-names>R</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Just a flexible linker? the structural and dynamic properties of CBP-ID4 Revealed by NMR Spectroscopy</article-title><source>Biophysical Journal</source><volume>110</volume><fpage>372</fpage><lpage>381</lpage><pub-id pub-id-type="doi">10.1016/j.bpj.2015.11.3516</pub-id></element-citation></ref><ref id="bib52"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Qin</surname><given-names>S</given-names></name><name><surname>Hicks</surname><given-names>A</given-names></name><name><surname>Dey</surname><given-names>S</given-names></name><name><surname>Prasad</surname><given-names>R</given-names></name><name><surname>Zhou</surname><given-names>HX</given-names></name></person-group><year iso-8601-date="2022">2022</year><article-title>ReSMAP: web server for predicting residue-specific membrane-association propensities of intrinsically disordered proteins</article-title><source>Membranes</source><volume>12</volume><elocation-id>773</elocation-id><pub-id pub-id-type="doi">10.3390/membranes12080773</pub-id><pub-id pub-id-type="pmid">36005688</pub-id></element-citation></ref><ref id="bib53"><element-citation publication-type="web"><person-group person-group-type="author"><name><surname>Qin</surname><given-names>S</given-names></name><name><surname>Zhou</surname><given-names>HX</given-names></name></person-group><year iso-8601-date="2024">2024</year><article-title>SeqDYN: Predictor for the Sequence-Dependent Backbone Dynamics of Intrinsically Disordered Proteins</article-title><ext-link ext-link-type="uri" xlink:href="https://zhougroup-uic.github.io/SeqDYNidp/">https://zhougroup-uic.github.io/SeqDYNidp/</ext-link><date-in-citation iso-8601-date="2024-10-23">October 23, 2024</date-in-citation></element-citation></ref><ref id="bib54"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Rezaei-Ghaleh</surname><given-names>N</given-names></name><name><surname>Parigi</surname><given-names>G</given-names></name><name><surname>Soranno</surname><given-names>A</given-names></name><name><surname>Holla</surname><given-names>A</given-names></name><name><surname>Becker</surname><given-names>S</given-names></name><name><surname>Schuler</surname><given-names>B</given-names></name><name><surname>Luchinat</surname><given-names>C</given-names></name><name><surname>Zweckstetter</surname><given-names>M</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Local and global dynamics in intrinsically disordered synuclein</article-title><source>Angewandte Chemie</source><volume>57</volume><fpage>15262</fpage><lpage>15266</lpage><pub-id pub-id-type="doi">10.1002/anie.201808172</pub-id><pub-id pub-id-type="pmid">30184304</pub-id></element-citation></ref><ref id="bib55"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Rezaei-Ghaleh</surname><given-names>N</given-names></name><name><surname>Parigi</surname><given-names>G</given-names></name><name><surname>Zweckstetter</surname><given-names>M</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>reorientational dynamics of amyloid-β from NMR spin relaxation and molecular simulation</article-title><source>The Journal of Physical Chemistry Letters</source><volume>10</volume><fpage>3369</fpage><lpage>3375</lpage><pub-id pub-id-type="doi">10.1021/acs.jpclett.9b01050</pub-id><pub-id pub-id-type="pmid">31181936</pub-id></element-citation></ref><ref id="bib56"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Rives</surname><given-names>A</given-names></name><name><surname>Meier</surname><given-names>J</given-names></name><name><surname>Sercu</surname><given-names>T</given-names></name><name><surname>Goyal</surname><given-names>S</given-names></name><name><surname>Lin</surname><given-names>Z</given-names></name><name><surname>Liu</surname><given-names>J</given-names></name><name><surname>Guo</surname><given-names>D</given-names></name><name><surname>Ott</surname><given-names>M</given-names></name><name><surname>Zitnick</surname><given-names>CL</given-names></name><name><surname>Ma</surname><given-names>J</given-names></name><name><surname>Fergus</surname><given-names>R</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>Biological structure and function emerge from scaling unsupervised learning to 250 million protein sequences</article-title><source>PNAS</source><volume>118</volume><elocation-id>e2016239118</elocation-id><pub-id pub-id-type="doi">10.1073/pnas.2016239118</pub-id><pub-id pub-id-type="pmid">33876751</pub-id></element-citation></ref><ref id="bib57"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Robustelli</surname><given-names>P</given-names></name><name><surname>Ibanez-de-Opakua</surname><given-names>A</given-names></name><name><surname>Campbell-Bezat</surname><given-names>C</given-names></name><name><surname>Giordanetto</surname><given-names>F</given-names></name><name><surname>Becker</surname><given-names>S</given-names></name><name><surname>Zweckstetter</surname><given-names>M</given-names></name><name><surname>Pan</surname><given-names>AC</given-names></name><name><surname>Shaw</surname><given-names>DE</given-names></name></person-group><year iso-8601-date="2022">2022</year><article-title>Molecular basis of small-molecule binding to α-Synuclein</article-title><source>Journal of the American Chemical Society</source><volume>144</volume><fpage>2501</fpage><lpage>2510</lpage><pub-id pub-id-type="doi">10.1021/jacs.1c07591</pub-id><pub-id pub-id-type="pmid">35130691</pub-id></element-citation></ref><ref id="bib58"><element-citation publication-type="software"><person-group person-group-type="author"><name><surname>Rohatgi</surname><given-names>A</given-names></name></person-group><year iso-8601-date="2022">2022</year><data-title>Webplotdigitizer</data-title><version designator="4.6">4.6</version><publisher-name>WebPlotDigitizer</publisher-name><ext-link ext-link-type="uri" xlink:href="https://apps.automeris.io/wpd4/">https://apps.automeris.io/wpd4/</ext-link></element-citation></ref><ref id="bib59"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Schiavina</surname><given-names>M</given-names></name><name><surname>Salladini</surname><given-names>E</given-names></name><name><surname>Murrali</surname><given-names>MG</given-names></name><name><surname>Tria</surname><given-names>G</given-names></name><name><surname>Felli</surname><given-names>IC</given-names></name><name><surname>Pierattelli</surname><given-names>R</given-names></name><name><surname>Longhi</surname><given-names>S</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>Ensemble description of the intrinsically disordered N-terminal domain of the Nipah virus P/V protein from combined NMR and SAXS</article-title><source>Scientific Reports</source><volume>10</volume><elocation-id>19574</elocation-id><pub-id pub-id-type="doi">10.1038/s41598-020-76522-3</pub-id><pub-id pub-id-type="pmid">33177626</pub-id></element-citation></ref><ref id="bib60"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Schneider</surname><given-names>R</given-names></name><name><surname>Maurin</surname><given-names>D</given-names></name><name><surname>Communie</surname><given-names>G</given-names></name><name><surname>Kragelj</surname><given-names>J</given-names></name><name><surname>Hansen</surname><given-names>DF</given-names></name><name><surname>Ruigrok</surname><given-names>RWH</given-names></name><name><surname>Jensen</surname><given-names>MR</given-names></name><name><surname>Blackledge</surname><given-names>M</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>Visualizing the molecular recognition trajectory of an intrinsically disordered protein using multinuclear relaxation dispersion NMR</article-title><source>Journal of the American Chemical Society</source><volume>137</volume><fpage>1220</fpage><lpage>1229</lpage><pub-id pub-id-type="doi">10.1021/ja511066q</pub-id><pub-id pub-id-type="pmid">25551399</pub-id></element-citation></ref><ref id="bib61"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Schwalbe</surname><given-names>H</given-names></name><name><surname>Fiebig</surname><given-names>KM</given-names></name><name><surname>Buck</surname><given-names>M</given-names></name><name><surname>Jones</surname><given-names>JA</given-names></name><name><surname>Grimshaw</surname><given-names>SB</given-names></name><name><surname>Spencer</surname><given-names>A</given-names></name><name><surname>Glaser</surname><given-names>SJ</given-names></name><name><surname>Smith</surname><given-names>LJ</given-names></name><name><surname>Dobson</surname><given-names>CM</given-names></name></person-group><year iso-8601-date="1997">1997</year><article-title>Structural and dynamical properties of a denatured protein. Heteronuclear 3D NMR experiments and theoretical simulations of lysozyme in 8 M urea</article-title><source>Biochemistry</source><volume>36</volume><fpage>8977</fpage><lpage>8991</lpage><pub-id pub-id-type="doi">10.1021/bi970049q</pub-id><pub-id pub-id-type="pmid">9220986</pub-id></element-citation></ref><ref id="bib62"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Schwarzinger</surname><given-names>S</given-names></name><name><surname>Wright</surname><given-names>PE</given-names></name><name><surname>Dyson</surname><given-names>HJ</given-names></name></person-group><year iso-8601-date="2002">2002</year><article-title>Molecular hinges in protein folding: the urea-denatured state of apomyoglobin</article-title><source>Biochemistry</source><volume>41</volume><fpage>12681</fpage><lpage>12686</lpage><pub-id pub-id-type="doi">10.1021/bi020381o</pub-id><pub-id pub-id-type="pmid">12379110</pub-id></element-citation></ref><ref id="bib63"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Sekiyama</surname><given-names>N</given-names></name><name><surname>Takaba</surname><given-names>K</given-names></name><name><surname>Maki-Yonekura</surname><given-names>S</given-names></name><name><surname>Akagi</surname><given-names>K-I</given-names></name><name><surname>Ohtani</surname><given-names>Y</given-names></name><name><surname>Imamura</surname><given-names>K</given-names></name><name><surname>Terakawa</surname><given-names>T</given-names></name><name><surname>Yamashita</surname><given-names>K</given-names></name><name><surname>Inaoka</surname><given-names>D</given-names></name><name><surname>Yonekura</surname><given-names>K</given-names></name><name><surname>Kodama</surname><given-names>TS</given-names></name><name><surname>Tochio</surname><given-names>H</given-names></name></person-group><year iso-8601-date="2022">2022</year><article-title>ALS mutations in the TIA-1 prion-like domain trigger highly condensed pathogenic structures</article-title><source>PNAS</source><volume>119</volume><elocation-id>e2122523119</elocation-id><pub-id pub-id-type="doi">10.1073/pnas.2122523119</pub-id><pub-id pub-id-type="pmid">36112647</pub-id></element-citation></ref><ref id="bib64"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Shen</surname><given-names>Y</given-names></name><name><surname>Delaglio</surname><given-names>F</given-names></name><name><surname>Cornilescu</surname><given-names>G</given-names></name><name><surname>Bax</surname><given-names>A</given-names></name></person-group><year iso-8601-date="2009">2009</year><article-title>TALOS+: a hybrid method for predicting protein backbone torsion angles from NMR chemical shifts</article-title><source>Journal of Biomolecular NMR</source><volume>44</volume><fpage>213</fpage><lpage>223</lpage><pub-id pub-id-type="doi">10.1007/s10858-009-9333-z</pub-id><pub-id pub-id-type="pmid">19548092</pub-id></element-citation></ref><ref id="bib65"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Smrt</surname><given-names>ST</given-names></name><name><surname>Escobar</surname><given-names>CA</given-names></name><name><surname>Dey</surname><given-names>S</given-names></name><name><surname>Cross</surname><given-names>TA</given-names></name><name><surname>Zhou</surname><given-names>HX</given-names></name></person-group><year iso-8601-date="2023">2023</year><article-title>An Arg/Ala-rich helix in the N-terminal region of M. tuberculosis FtsQ is a potential membrane anchor of the Z-ring</article-title><source>Communications Biology</source><volume>6</volume><elocation-id>311</elocation-id><pub-id pub-id-type="doi">10.1038/s42003-023-04686-5</pub-id><pub-id pub-id-type="pmid">36959324</pub-id></element-citation></ref><ref id="bib66"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Sólyom</surname><given-names>Z</given-names></name><name><surname>Ma</surname><given-names>P</given-names></name><name><surname>Schwarten</surname><given-names>M</given-names></name><name><surname>Bosco</surname><given-names>M</given-names></name><name><surname>Polidori</surname><given-names>A</given-names></name><name><surname>Durand</surname><given-names>G</given-names></name><name><surname>Willbold</surname><given-names>D</given-names></name><name><surname>Brutscher</surname><given-names>B</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>The disordered region of the HCV Protein NS5A: conformational dynamics, SH3 binding, and phosphorylation</article-title><source>Biophysical Journal</source><volume>109</volume><fpage>1483</fpage><lpage>1496</lpage><pub-id pub-id-type="doi">10.1016/j.bpj.2015.06.040</pub-id><pub-id pub-id-type="pmid">26445449</pub-id></element-citation></ref><ref id="bib67"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Song</surname><given-names>J</given-names></name><name><surname>Guo</surname><given-names>L-W</given-names></name><name><surname>Muradov</surname><given-names>H</given-names></name><name><surname>Artemyev</surname><given-names>NO</given-names></name><name><surname>Ruoho</surname><given-names>AE</given-names></name><name><surname>Markley</surname><given-names>JL</given-names></name></person-group><year iso-8601-date="2008">2008</year><article-title>Intrinsically disordered gamma-subunit of cGMP phosphodiesterase encodes functionally relevant transient secondary and tertiary structure</article-title><source>PNAS</source><volume>105</volume><fpage>1505</fpage><lpage>1510</lpage><pub-id pub-id-type="doi">10.1073/pnas.0709558105</pub-id><pub-id pub-id-type="pmid">18230733</pub-id></element-citation></ref><ref id="bib68"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Sung</surname><given-names>YH</given-names></name><name><surname>Eliezer</surname><given-names>D</given-names></name></person-group><year iso-8601-date="2007">2007</year><article-title>Residual structure, backbone dynamics, and interactions within the synuclein family</article-title><source>Journal of Molecular Biology</source><volume>372</volume><fpage>689</fpage><lpage>707</lpage><pub-id pub-id-type="doi">10.1016/j.jmb.2007.07.008</pub-id><pub-id pub-id-type="pmid">17681534</pub-id></element-citation></ref><ref id="bib69"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Szalainé Ágoston</surname><given-names>B</given-names></name><name><surname>Kovács</surname><given-names>D</given-names></name><name><surname>Tompa</surname><given-names>P</given-names></name><name><surname>Perczel</surname><given-names>A</given-names></name></person-group><year iso-8601-date="2011">2011</year><article-title>Full backbone assignment and dynamics of the intrinsically disordered dehydrin ERD14</article-title><source>Biomolecular NMR Assignments</source><volume>5</volume><fpage>189</fpage><lpage>193</lpage><pub-id pub-id-type="doi">10.1007/s12104-011-9297-2</pub-id></element-citation></ref><ref id="bib70"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Tesei</surname><given-names>G</given-names></name><name><surname>Lindorff-Larsen</surname><given-names>K</given-names></name></person-group><year iso-8601-date="2022">2022</year><article-title>Improved predictions of phase behaviour of intrinsically disordered proteins by tuning the interaction range</article-title><source>Open Research Europe</source><volume>2</volume><elocation-id>94</elocation-id><pub-id pub-id-type="doi">10.12688/openreseurope.14967.2</pub-id><pub-id pub-id-type="pmid">37645312</pub-id></element-citation></ref><ref id="bib71"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Thapa</surname><given-names>C</given-names></name><name><surname>Roivas</surname><given-names>P</given-names></name><name><surname>Haataja</surname><given-names>T</given-names></name><name><surname>Permi</surname><given-names>P</given-names></name><name><surname>Pentikäinen</surname><given-names>U</given-names></name></person-group><year iso-8601-date="2022">2022</year><article-title>Interaction mechanism of endogenous PP2A inhibitor protein ENSA with PP2A</article-title><source>The FEBS Journal</source><volume>289</volume><fpage>519</fpage><lpage>534</lpage><pub-id pub-id-type="doi">10.1111/febs.16150</pub-id><pub-id pub-id-type="pmid">34346186</pub-id></element-citation></ref><ref id="bib72"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Thapar</surname><given-names>R</given-names></name><name><surname>Mueller</surname><given-names>GA</given-names></name><name><surname>Marzluff</surname><given-names>WF</given-names></name></person-group><year iso-8601-date="2004">2004</year><article-title>The N-terminal domain of the <italic>Drosophila</italic> histone mRNA binding protein, SLBP, Is Intrinsically disordered with nascent helical structure <sup>,</sup></article-title><source>Biochemistry</source><volume>43</volume><fpage>9390</fpage><lpage>9400</lpage><pub-id pub-id-type="doi">10.1021/bi036314r</pub-id></element-citation></ref><ref id="bib73"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Vogel</surname><given-names>A</given-names></name><name><surname>Crawford</surname><given-names>A</given-names></name><name><surname>Nyarko</surname><given-names>A</given-names></name></person-group><year iso-8601-date="2022">2022</year><article-title>Multivalent Angiomotin-like 1 and Yes-associated protein form a dynamic complex</article-title><source>Protein Science</source><volume>31</volume><elocation-id>e4295</elocation-id><pub-id pub-id-type="doi">10.1002/pro.4295</pub-id><pub-id pub-id-type="pmid">35481651</pub-id></element-citation></ref><ref id="bib74"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname><given-names>X</given-names></name><name><surname>Zhang</surname><given-names>S</given-names></name><name><surname>Zhang</surname><given-names>J</given-names></name><name><surname>Huang</surname><given-names>X</given-names></name><name><surname>Xu</surname><given-names>C</given-names></name><name><surname>Wang</surname><given-names>W</given-names></name><name><surname>Liu</surname><given-names>Z</given-names></name><name><surname>Wu</surname><given-names>J</given-names></name><name><surname>Shi</surname><given-names>Y</given-names></name></person-group><year iso-8601-date="2010">2010</year><article-title>A large intrinsically disordered region in SKIP and its disorder-order transition induced by PPIL1 binding revealed by NMR</article-title><source>The Journal of Biological Chemistry</source><volume>285</volume><fpage>4951</fpage><lpage>4963</lpage><pub-id pub-id-type="doi">10.1074/jbc.M109.087528</pub-id><pub-id pub-id-type="pmid">20007319</pub-id></element-citation></ref><ref id="bib75"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname><given-names>J</given-names></name><name><surname>Choi</surname><given-names>J-M</given-names></name><name><surname>Holehouse</surname><given-names>AS</given-names></name><name><surname>Lee</surname><given-names>HO</given-names></name><name><surname>Zhang</surname><given-names>X</given-names></name><name><surname>Jahnel</surname><given-names>M</given-names></name><name><surname>Maharana</surname><given-names>S</given-names></name><name><surname>Lemaitre</surname><given-names>R</given-names></name><name><surname>Pozniakovsky</surname><given-names>A</given-names></name><name><surname>Drechsel</surname><given-names>D</given-names></name><name><surname>Poser</surname><given-names>I</given-names></name><name><surname>Pappu</surname><given-names>RV</given-names></name><name><surname>Alberti</surname><given-names>S</given-names></name><name><surname>Hyman</surname><given-names>AA</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>A molecular grammar governing the driving forces for phase separation of prion-like RNA binding proteins</article-title><source>Cell</source><volume>174</volume><fpage>688</fpage><lpage>699</lpage><pub-id pub-id-type="doi">10.1016/j.cell.2018.06.006</pub-id><pub-id pub-id-type="pmid">29961577</pub-id></element-citation></ref><ref id="bib76"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Wirmer</surname><given-names>J</given-names></name><name><surname>Peti</surname><given-names>W</given-names></name><name><surname>Schwalbe</surname><given-names>H</given-names></name></person-group><year iso-8601-date="2006">2006</year><article-title>Motional properties of unfolded ubiquitin: a model for a random coil protein</article-title><source>Journal of Biomolecular NMR</source><volume>35</volume><fpage>175</fpage><lpage>186</lpage><pub-id pub-id-type="doi">10.1007/s10858-006-9026-9</pub-id><pub-id pub-id-type="pmid">16865418</pub-id></element-citation></ref><ref id="bib77"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Wong</surname><given-names>LE</given-names></name><name><surname>Kim</surname><given-names>TH</given-names></name><name><surname>Muhandiram</surname><given-names>DR</given-names></name><name><surname>Forman-Kay</surname><given-names>JD</given-names></name><name><surname>Kay</surname><given-names>LE</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>NMR experiments for studies of dilute and condensed protein phases: application to the phase-separating protein CAPRIN1</article-title><source>Journal of the American Chemical Society</source><volume>142</volume><fpage>2471</fpage><lpage>2489</lpage><pub-id pub-id-type="doi">10.1021/jacs.9b12208</pub-id><pub-id pub-id-type="pmid">31898464</pub-id></element-citation></ref><ref id="bib78"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Xie</surname><given-names>M</given-names></name><name><surname>Li</surname><given-names>D</given-names></name><name><surname>Yuan</surname><given-names>J</given-names></name><name><surname>Hansen</surname><given-names>AL</given-names></name><name><surname>Brüschweiler</surname><given-names>R</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Quantitative binding behavior of intrinsically disordered proteins to nanoparticle surfaces at individual residue level</article-title><source>Chemistry – A European Journal</source><volume>24</volume><fpage>16997</fpage><lpage>17001</lpage><pub-id pub-id-type="doi">10.1002/chem.201804556</pub-id></element-citation></ref><ref id="bib79"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Yao</surname><given-names>S</given-names></name><name><surname>Lee</surname><given-names>EF</given-names></name><name><surname>Pettikiriarachchi</surname><given-names>A</given-names></name><name><surname>Evangelista</surname><given-names>M</given-names></name><name><surname>Keizer</surname><given-names>DW</given-names></name><name><surname>Fairlie</surname><given-names>WD</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Characterisation of the conformational preference and dynamics of the intrinsically disordered N-terminal region of Beclin 1 by NMR spectroscopy</article-title><source>Biochimica et Biophysica Acta</source><volume>1864</volume><fpage>1128</fpage><lpage>1137</lpage><pub-id pub-id-type="doi">10.1016/j.bbapap.2016.06.005</pub-id><pub-id pub-id-type="pmid">27288992</pub-id></element-citation></ref><ref id="bib80"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Yu</surname><given-names>L</given-names></name><name><surname>Brüschweiler</surname><given-names>R</given-names></name></person-group><year iso-8601-date="2022">2022</year><article-title>Quantitative prediction of ensemble dynamics, shapes and contact propensities of intrinsically disordered proteins</article-title><source>PLOS Computational Biology</source><volume>18</volume><elocation-id>e1010036</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pcbi.1010036</pub-id></element-citation></ref><ref id="bib81"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname><given-names>Y</given-names></name><name><surname>Prasad</surname><given-names>R</given-names></name><name><surname>Su</surname><given-names>S</given-names></name><name><surname>Lee</surname><given-names>D</given-names></name><name><surname>Zhou</surname><given-names>HX</given-names></name></person-group><year iso-8601-date="2024">2024</year><article-title>Amino acid-dependent phase equilibrium and material properties of tetrapeptide condensates</article-title><source>Cell Reports Physical Science</source><volume>5</volume><elocation-id>102218</elocation-id><pub-id pub-id-type="doi">10.1016/j.xcrp.2024.102218</pub-id></element-citation></ref><ref id="bib82"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Zheng</surname><given-names>Z</given-names></name><name><surname>Ma</surname><given-names>D</given-names></name><name><surname>Yahr</surname><given-names>TL</given-names></name><name><surname>Chen</surname><given-names>L</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>The transiently ordered regions in intrinsically disordered ExsE are correlated with structural elements involved in chaperone binding</article-title><source>Biochemical and Biophysical Research Communications</source><volume>417</volume><fpage>129</fpage><lpage>134</lpage><pub-id pub-id-type="doi">10.1016/j.bbrc.2011.11.070</pub-id><pub-id pub-id-type="pmid">22138394</pub-id></element-citation></ref><ref id="bib83"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Zimmerman</surname><given-names>JM</given-names></name><name><surname>Eliezer</surname><given-names>N</given-names></name><name><surname>Simha</surname><given-names>R</given-names></name></person-group><year iso-8601-date="1968">1968</year><article-title>The characterization of amino acid sequences in proteins by statistical methods</article-title><source>Journal of Theoretical Biology</source><volume>21</volume><fpage>170</fpage><lpage>201</lpage><pub-id pub-id-type="doi">10.1016/0022-5193(68)90069-6</pub-id><pub-id pub-id-type="pmid">5700434</pub-id></element-citation></ref></ref-list><app-group><app id="appendix-1"><title>Appendix 1</title><sec sec-type="appendix" id="s8"><title>Sequences of 54 IDPs</title><p>Terminal tags and other insertions are underlined.</p><sec sec-type="appendix" id="s8-1"><title>Training set (45 IDPs)</title><p><code xml:space="preserve">&gt;A1-LCD (Uniprot ID P04256, residues 186-320; deletion of 258-263)
GSMASASSSQRGRSGSGNFGGGRGGGFGGNDNFGRGGNFSGRGGFGGSRGGGGYGGSGDGYNGFGNDGSNFGGGGNYNNQSSNFGPMKGGNFGGRSSGPYGGGGQYFAKPRNQGGYGGSSSSSSYGSGRRF
&gt;Aβ40 (Uniprot ID Q28053, residues 7-46)
DAEFRHDSGYEVHHQKLVFFAEDVGSNKGAIIGLMVGGV
&gt;Ash1 (Uniprot ID P34233, residues 420-500)
GASASSSPSPSTPTKSGKMRSRSSSPVRPKAYTPSPRSPNYHRFALDSPPQSPRRSSNSSITKKGSRRSSGSSPTRHTTRVCV
&gt;Beclin1 (Uniprot ID Q14457, residues 1-150)
MGSSHHHHHHSQDPMEGSKTSNNSTMQVSFVSQRSSQPLKLDTSFKILDRVTIQELTAPLLTTAQAKPGETQEEETNSGEEPFIETPRQDGVSRRFIPPARMMSTESANSFTLIGEASDGGTMENLSRRLKVTGDLFDIMSGQTDVDHPLSEESTDTLLDQLDTY
&gt;CAPRIN1 (Uniprot ID Q14444, residues 607-709)
SRGVSRGGSRGARGLMNGYRGPANGFRGGYDGYRPSFSNTPNSGYTQSQFSAPRDYSGYQRDGYQQNFKRGSGQSGPRGAPRGRGGPPRPNRGMPQMNTQQVN
&gt;CBP-ID4 (Uniprot ID Q92793, residues 1852-2057)
MQQQIQHRLQQAQLMRRRMATMNTRNVPQQSLPSPTSAPPGTPTQQPSTPQTPQPPAQPQPSPVSMSPAGFPSVARTQPPTTVSTGKPTSQVPAPPPPAQPPPAAVEAARQIEREAQQQQHLYRVNINNSMPPGRTGMGTPGSQMAPVSLNVPRPNQVSGPVMPSMPPGQWQQAPLPQQQPMPGLPRPVISMQAQAAVAGPRMPSVQ
&gt;GbnD4-DHD (Uniprot ID A0A808VWJ6, residues 412-482)
MKHHHHHHHHGGLVPRGSHGSDEGVPDALRADTVPRAGPVRYARRRYWIGEARSDALAPAAPLEREPLPAEAMGAYFAIRRTDADDTVAAH
&gt;ERD14 (Uniprot ID P42763, residues 1-185)
MAEEIKNVPEQEVPKVATEESSAEVTDRGLFDFLGKKKDETKPEETPIASEFEQKVHISEPEPEVKHESLLEKLHRSDSSSSSSSEEEGSDGEKRKKKKEKKKPTTEVEVKEEEKKGFMEKLKEKLPGHKKPEDGSAVAAAPVVVPPPVEEAHPVEKKGILEKIKEKLPGYHPKTTVEEEKKDKE
&gt;ExsE (Uniprot ID Q9I322, residues 1-81)
MKIESIPPVQPSQDAGAEAVGHFEGRSVTRAAVRGDDRSSVAGLARWLARNVAGDPRSEQALQRLADGDGTPLEARTVRRREFLEGSS
&gt;FCP1 (Uniprot ID Q9Y5B0, residues 879-961)
PGPEEQEEEPQPRKPGTRRERTLGAPASSERSAAGGRGPRGHKRKLNEEDAASESSRESSNEDEGSSSEADEMAKALEAELNDLM
&gt;FUS (Uniprot ID P35637, residues 1-163)
MASNDYTQQATQSYGAYPTQPGQGYSQQSSQPYGQQSYSGYSQSTDTSGYGQSSYSSYGQSQNTGYGTQSTPQGYGSTGGYGSSQSSQSSYGQQSSYPGYGQQPAPSSTSGSYGSSSQSSSYGQPQSGSYSQQPSYGGQQQSYGQQQSYNPPQGYGQQNQYNS
&gt;GAb1 (Uniprot ID B7Z3B9, residues 510-591)
SSPMIKPKGDKQVEYLDLDLDSGKSTPPRKQKSSGSGSSVADERVDYVVVDQQKTLALKSTREAWTDGRQSTESETPAKSVK
&gt;hACTR (Uniprot ID Q9Y6Q9, residues 1023-1091)
GTQNRPLLRNSLDDLVGPPSNLEGQSDERALLDQLHTLLSNTDATGLEEIDRALGIPELVNQGQALEPK&gt;Hahellin (Uniprot ID Q2SHN6, residues 162-252)
MGEKTVKLYEDTHFKGYSVELPVGDYNLSSLISRGALNDDLSSARVPSGLRLEVFQHNNFKGVRDFYTSDAAELSRDNDASSVRVSKMETTN
&gt;hCSD1 (Uniprot ID P20810, residues 137-277)
AVPVESKPDKPSGKSGMDAALDDLIDTLGGPEETEEENTTYTGPEVSDPMSSTYIEELGKREVTIPPKYRELLAKKEGITGPPADSSKPIGPDDAIDALSSDFTCGSPTAAGKKTEKEESTEVLKAQSAGTVRSAAPPQEK
&gt;HOX-DFD (Uniprot ID P07548, residues 337-426)
TDGERIIYPWMKKIHVAGVANGSYQPGMEPKRQRTAYTRHQILELEKEFHYNRYLTRRRRIEIAHTLVLSERQIKIWFQNRRMKWKKDNK
&gt;hZIP4-ICL2 (Uniprot ID Q6P5W5, residues 424-498)
GDRGPEFELGTLPRDPEDLEDGPCGHSSHSHGGHSHGVSLQLAPSELRQPKPPHEGSRADLVAEESPELLNPEPRRLSPELRLLPYGHGLSAWSHPQFEK
&gt;Jaburetox (Uniprot ID I1K3K3, residues 230-320)
MGPVNEANCKAAMEIVCRREFGHKEEEDASEGVTTGDPDCPFTKAIPREEYANKYGPTIGDKIRLGDTDLIAEIEKDFALYGDESVFGGGKVIH
&gt;KRS-NT (Uniprot ID Q15046, residues 1-72)
MAAVQAAEVKVDGSEPKLSKNELKRRLKAEKKVAEKEAKQKELSEKQLSQATAAATNHTTDNGVGPEEESVD
&gt;MBP-xα2 (Uniprot ID P04370, residues 172-237)
SIGRFFSGDRGAPKRGSGKDSHTRTTHYGSLPQKSQHGRTQDENPVVHFFKNIVTPRTPPPSQGKGRGLS
&gt;MKK4 (Uniprot ID P45985, residues 1-86)
MAAPSPSGGGGSGGGSGSGTPGPVGSPAPGHPAVSSMQGKRKALKLNFANPPFKSTARFTLNPNPTGVQNPHIERLRTHSIESSGK
&gt;N-Cby (Uniprot ID B0QY54, residues 1-63)
MPFFGNTFSPKKTPPRKSASLSNLHSLDRSTREVELGLEYGSPTMNLAGQSLKFENGQWIAET
&gt;Niv-PNTD (Uniprot ID P0C1C7, residues 1-406)
MDKLELVNDGLNIIDFIQKNQKEIQKTYGRSSIQQPSIKDQTKAWEDFLQCTSGESEQVEGGMSKDDGDVERRNLEDLSSTSPTDGTIGKRVSNTRDWAEGSDDIQLDPVVTDVVYHDHGGECTGYGFTSSPERGWSDYTSGANNGNVCLVSDAKMLSYAPEIAVSKEDRETDLVHLENKLSTTGLNPTAVPFTLRNLSDPAKDSPVIAEHYYGLGVKEQNVGPQTSRNVNLDSIKLYTSDDEEADQLEFEDEFAGSSSEVIVGISPEDEEPSSVGGKPNESIGRTIEGQSIRDNLQAKDNKSTDVPGAGPKDSAVKEEPPQKRLPMLAEEFECSGSEDPIIRELLKENSLINCQQGKDAQPPYHWSIERSISPDKTEIVNGAVQTADRQRPGTPMPKSRGIPIKK
&gt;NS5A-D2D3 (Uniprot ID O92972, residues 2163-2419)
GHMASGSLRGGEPEPDVTVLTSMLTDPSHITAETAKRRLARGSPPSLASSSASQLSAPSLKATCTTHHDSPDADLIEANLLWRQEMGGNITRVESENKVVILDSFEPLHADGDEREISVAAEILRKSRKFPSALPIWARPDYNPPLLESWKDPDYVPPVVHGCPLPPTKAPPIPPPRRKRTVVLTESNVSSALAELATKTFGSSGSSAVDSGTATALPDQASDDGDKGSDVESYSSMPPLEGEPGDPDLSDGSWSTVSEEASEDVVCC
&gt;NUPR1 (Uniprot ID O60356, residues 2-82)
MRGSHHHHHHGSATFPPATSAPQQPPGPEDEDSSLDESDLYSLAHSYLGGGGRKGRTKREAAANTNRPSPGGHERKLVTKLQNSERKKRGARR
OPN (Uniprot ID F1NSM8, residues 46-264)
MHQDHVDSQSQEHLQQTQNDLASLQQTHYSSEENADVPEQPDFPDVPSKSQETVDDDDDDDNDSNDTDESDEVFTDFPTEAPVAPFNRGDNAGRGDSVAYGFRAKAHVVKASKIRKAARKLIEDDATTEDGDSQPAGLWWPKESREQNSRELPQHQSVENDSRPKFDSREVDGGDSKASAGVDSRESQGSVPAVDASNQTLESAEDAEDRHSIENNEVTR
&gt;p53TAD (Uniprot ID P04637, residues 1-71)
MEEPQSDPSVEPPLSQETFSDLWKLLPENNVLSPLPSQAMDDLMLSPDDIEQWFTEDPGPDEAPRMPEAAPRV
&gt;PDEγ (Uniprot ID P61248, residues 1-87)
MNLEPPKAEIRSATRVMGGPVTPRKGPPKFKQRQTRQFKSKPPKKGVQGFGDDIPGMEGLGTDITVIAPWEAFNHLELHELAQYGII
&gt;PKIα (Uniprot ID P61925, residues 2-76)
TDVETTYADFIASGRTGRRNAIHDILVSSASGNSNELALKLAGLDINKTEGEEDAQRSSTEQSGEAQGEAAKSES
&gt;Mev-PNTD (Uniprot ID P03422, residues 1-304)
MAEEQARHVKNGLECIRALKAEPIGSLAIEEAMAAWSEISDNPGQERATCREEKAGSSGLSKPCLSAIGSTEGGAPRIRGQGPGESDDDAETLGIPPRNLQASSTGLQCYYVYDHSGEAVKGIQDADSIMVQSGLDGDSTLSGGDNESENSDVDIGEPDTEGYAITDRGSAPISMGFRASDVETAEGGEIHELLRLQSRGNNFPKLGKTLNVPPPPDPGRASTSGTPIKKGTERRLASFGTEIASLLTGGATQCARKSPSEPSGPGAPAGNVPECVSNAALIQEWTPESGTTISPRSQNNEEGG
&gt;ProTα (Uniprot ID P06454, residues 1-111)
GPMSDAAVDTSSEITTKDLKEKKEVVEEAENGRDAPANGNAENEENGEQEADNEVDEEEEEGGEEEEEEEEGDGEEEDGDEDEEAESATGKRAAEDDEDDDVDTKKQKTDEDD
&gt;Pup (Uniprot ID P9WHN4, residues 1-64)
MAQEQTKRGGGGGDDDDIAGSTAAGQERREKLTEETDDLLDEIDDVLEENAEDFVRAYVQKGGQ
&gt;rmBG21 (Uniprot ID P04370, residues 2-190)
GNHSGKRELSAEKASKDGEIHRGEAGKKRSVGKLSQTASEDSDVFGEADAIQNNGTSAEDTAVTDSKHTADPKNNWQGAHPADPGNRPHLIRLFSRDAPGREDNTFKDRPSESDELQTIQEDPTAASGGLDVMASQKRPSQRSKYLATASTMDHARHGFLPRHRDTGILDSIGRFFSGDRGAPKRGSGKVSLEHHHHHH
&gt;RPB1 (Uniprot ID P24928, residues 1773-1970)
GHMSPNYTPTSPNYSPTSPSYSPTSPSYSPTSPSYSPSSPRYTPQSPTYTPSSPSYSPSSPSYSPASPKYTPTSPSYSPSSPEYTPTSPKYSPTSPKYSPTSPKYSPTSPTYSPTTPKYSPTSPTYSPTSPVYTPTSPKYSPTSPTYSPTSPKYSPTSPTYSPTSPKGSTYSPTSPGYSPTSPTYSLTSPAISPDDSDEEN
&gt;securin (Uniprot ID O95997, residues 1-202)
MATLIYVDKENGEPGTRVVAKDGLKLGSGPSIKALDGRSQVSTPRFGKTFDAPPALPKATRKALGTVNRATEKSVKTKGPLKQKQPSFSAKKMTEKTVKAKSSVPASDDAYPEIEKFFPFNPLDFESFDLPEEHQIAHLPLSGVPLMILDEERELEKLFQLGPPSPVKMPSPPWESNLLQSPSSILSTLDVELPPVCCDIDI
&gt;Sev-NT (Uniprot ID Q07097, residues 401-524)
LSGGDGAYHEPTGGGAIEVALDNADIDLETEAHADQDARGWGGESGERWARQVSGGHFVTLHGAERLEEETNDEDVSDIERRIAMRLAERRQEDSATHGDEGRNNGVDHDEDDDAAAVAGIGGI
&gt;Sic1 (Uniprot ID P38634, residues 1-90)
GSMTPSTPPRSRGTRYLAQPSGNTSSSALMQGQKTPQKPSQNLVPVTPSTTKSFKNAPLLAPPNSNMGMTSPFNGLTSPQRSPFPKSSVKRT
&gt;SKIPN (Uniprot ID G3V5R3, residues 59-129)
GDGGAFPEIHVAQYPLDMGRKKKMSNALAIQVDSEGKIKYDAIARQGQSKDKVIYSKYTDLVPKEVMNADD
&gt;SLBP-NT (Uniprot ID Q9VAN6, residues 17-108)
MGSSHHHHHHSSGLVPRGSHMGSGSLNSSASSISIDVKPTMQSWAQEVRAEFGHSDEASSSLNSSAASCGSLAKKETADGNLESKDGEGREMAFEFLDGVNEVKFERLVKEEK
&gt;α-synuclein (Uniprot ID P37840, residues 1-140)
MDVFMKGLSKAKEGVVAAAEKTKQGVAEAAGKTKEGVLYVGSKTKEGVVHGVATVAEKTKEQVTNVGGAVVTGVTAVAQKTVEGAGSIAAATGFVKKDQLGKNEEGAPQEGILEDMPVDPDNEAYEMPSEEGYQDYEPEA
&gt;SOCS5-JIR (Uniprot ID A0A5E4BAI0, residues 12-81)
RSLRQRLQDTVGLCFPMRTYSKQSKPLFSNKRKIHLSELMLEKCPFPAGSDLAQKWHLIKQHTAPVSPHS
&gt;tau K18 (Uniprot ID Q9MYX8, residues 186-314)
QTAPVPMPDLKNVKSKIGSTENLKHQPGGGKVQIINKKLDLSNVQSKCGSKDNIKHVPGGGSVQIVYKPVDLSKVTSKAGSLGNIHHKPGGGQVEVKSEKLDFKDRVQSKIGSLDNITHVPGGGNKKIE
&gt;TC1 (Uniprot ID Q9NR00, residues 1-106)
MKAKRSHQAIIMSTSLRVSPSIHGYHFDTASRKKAVGNIFENTDQESLERLFRNSGDKKAEERAKIIFAIDQDVEEKTRALMALKKRTKDKLFQFLKLRKYSIKVH
&gt;TDP-43 (Uniprot ID Q13148, residues 267-414)
GHMNRQLERSGRFGGNPGGFGNQGGFGNSRGGGAGLGNNQGSNMGGGMNFGAFSINPAMMAAAQAALQSSWGMMGMLASQQNQSGPSGNNQNQGNMQREPNQAFGSGNNSYSGSNSGAAIGWGSASNAGSGSGFNGGFGSSMDSKSSGWGM
&gt;γ-tubulin-CT (Uniprot ID P53378, residues 439-473)
LLRGAAEQDSYLDDVLVDDENMVGELEEDLDADGDHKLV</code></p></sec><sec sec-type="appendix" id="s8-2"><title>Test set (9 IDPs)</title><p><code xml:space="preserve">&gt;AMOTL1 (Uniprot ID Q8IY63, residues 178-384)
STQPQQNNEELPTYEEAKAQSQFFRGQQQQQQQQGAVGHGYYMAGGTSQKSRTEGRPTVNRANSGQAHKDEALKELKQGHVRSLSERIMQLSLERNGAKQHLPGSGNGKGFKVGGGPSPAQPAGKVLDPRGPPPEYPFKTKQMMSPVSKTQEHGLFYGDQHPGMLHEMVKPYPAPQPVRTDVAVLRYQPPPEYGVTSRPCQLPFPST
&gt;CAHS-8 (Uniprot ID P0CU50, residues 1-227)
MSGRNVESHMERNEKVVVNNSGHADVKKQQQQVEHTEFTHTEVKAPLIHPAPPIISTGAAGLAEEIVGQGFTASAARISGGTAEVHLQPSAAMTEEARRDQERYRQEQESIAKQQEREMEKKTEAYRKTAEAEAEKIRKELEKQHARDVEFRKDLIESTIDRQKREVDLEAKMAKRELDREGQLAKEALERSRLATNVEVNFDSAAGHTVSGGTTVSTSDKMEIKRNENLYFQ
&gt;ChiZ (Uniprot ID I6YA32, residues 1-64)
MTPVRPPHTPDPLNLRGPLDGPRWRRAEPAQSRRPGRSRPGGAPLRYHRTGVGMSRTGHGSRPV
&gt;-endosulfine (Uniprot ID O43768, residues 1-121)
MSQKQEEENPAEETGEEKQDTQEKEGILPERAEEAKLKAKYPSLGQKPGGSDFLMKRLQKGQKYFDSGDYNMAKAKMKNKQLPSAGPDKNLVTGDHIPTPQDLPQRKSSLVTSKLAGGQVE
&gt;FtsQ (Uniprot ID Q8IY63, residues 1-99)
MTEHNEDPQIERVADDAADEEAVTEPLATESKDEPAEHPEFEGPRRRARRERAERRAAQARATAIEQARRAAKRRARGQIVSEQNPAKPAARGVVRGLK
&gt;Pdx1 (Uniprot ID P52945, residues 204-283)
GPGEEDKKRGGGTAVGGGGVAEPEQDCAVTSGEELLALPPPPPPGGAVPPAAPVAAREGRLPPGLSASPQPSSVAPRRPQEPR
&gt;synaptobrevin-2 (Uniprot ID P63027, residues 1-96)
MSATAATAPPAAPAGEGGPPAPPPNLTSNRRLQQTQAQVDEVVDIMRVNVDKVLERDQKLSELDDRADALQAGASQFETSAAKLKRKYWWKNLKMM
&gt;TIA-1 (Uniprot ID P31483, residues 320-386)
MGSSHHHHHHHHHHHHSENLYFQGGQYVPNGWQVPAYGVYGQPWSQQGFNQTQSSAPWMGPNYSVPPPQGQNGSMLPSQPAGYRVAGYETQ
&gt;YAP (Uniprot ID P46937, residues 50-171)
AGHQIVHVRGDSETDLEALFNAVMNPKTANVPQTVPMRLRKLPDSFFKPPEPKSHSRQASTDAGTAGALTPQHVRAHSSPASLQLGAVSPGTLTPTGVVSGPAATPTAQHLRQSSFEIPDDV</code></p></sec></sec></app></app-group></back><sub-article article-type="editor-report" id="sa0"><front-stub><article-id pub-id-type="doi">10.7554/eLife.88958.3.sa0</article-id><title-group><article-title>eLife assessment</article-title></title-group><contrib-group><contrib contrib-type="author"><name><surname>Cui</surname><given-names>Qiang</given-names></name><role specific-use="editor">Reviewing Editor</role><aff><institution>Boston University</institution><country>United States</country></aff></contrib></contrib-group><kwd-group kwd-group-type="evidence-strength"><kwd>Solid</kwd></kwd-group><kwd-group kwd-group-type="claim-importance"><kwd>Useful</kwd></kwd-group></front-stub><body><p>In this <bold>useful</bold> study, a <bold>solid</bold> machine learning approach based on a broad set of systems to predict the R2 relaxation rates of residues in intrinsically disordered proteins (IDPs) is described. The ability to predict the patterns of R2 will be helpful to guide experimental studies of IDPs. A potential weakness is that the predicted R2 values may include both fast and slow motions, thus the predictions provide only limited new physical insights into the nature of the underlying protein dynamics, such as the most relevant timescale.</p></body></sub-article><sub-article article-type="referee-report" id="sa1"><front-stub><article-id pub-id-type="doi">10.7554/eLife.88958.3.sa1</article-id><title-group><article-title>Reviewer #2 (Public review):</article-title></title-group><contrib-group><contrib contrib-type="author"><anonymous/><role specific-use="referee">Reviewer</role></contrib></contrib-group></front-stub><body><p>Qin, Sanbo and Zhou, Huan-Xiang created a model, SeqDYN, to predict nuclear magnetic resonance (NMR) spin relaxation spectra of intrinsically disordered proteins (IDPs), based primarily on amino acid sequence. To fit NMR data, SeqDYN uses 21 parameters, 20 that correspond to each amino acid, and a sequence correlation length for interactions. The model demonstrates that local sequence features impact the dynamics of the IDP, as SeqDYN performs better than a one residue predictor, despite having similar numbers of parameters. SeqDYN is trained using 45 IDP sequences and is retrained using both leave-one-out cross validation and five-fold cross validation, ensuring the model's robustness. While SeqDYN can provide reasonably accurate predictions in many cases, the authors note that improvements can be made by incorporating secondary structure predictions, especially for alpha-helices that exceed the correlation length of the model. The authors apply SeqDYN to study nine IDPs and a denatured ordered protein, demonstrating its predictive power. The model can be easily accessed via the website mentioned in the text.</p><p>The authors have adequately addressed the majority of my previous concerns. However, I still wonder if an attempt to fit the individual protein fitting parameter based on temperature and magnetic field strength would be possible. The authors would have 45 data points on which to fit such a parameter, which would only depend on two variables.</p></body></sub-article><sub-article article-type="referee-report" id="sa2"><front-stub><article-id pub-id-type="doi">10.7554/eLife.88958.3.sa2</article-id><title-group><article-title>Reviewer #3 (Public review):</article-title></title-group><contrib-group><contrib contrib-type="author"><anonymous/><role specific-use="referee">Reviewer</role></contrib></contrib-group></front-stub><body><p>The revised manuscript adds some new relevant analyses. It still, however, is unclear which timescales of motions the method refers to and there is confusion about whether the model can predict &quot;slower motions&quot;. While the authors answer some of my points, others are left unanswered. That is of course the authors' prerogative, and readers will in any case be able to read the reviewer comments. I am not sure it is productive to add further comments at this point.</p><p>Below are my comments from the first round of review:</p><p>The manuscript by Qin and Zhou presents an approach to predict dynamical properties of an intrinsically disordered protein (IDP) from sequence alone. In particular, the authors train a simple (but useful) machine learning model to predict (rescaled) NMR R2 values from sequence. Although these R2 rates only probe some aspects of IDR dynamics and the method does not provide insight into the molecular aspects of processes that lead to perturbed dynamics, the method can be useful to guide experiments.</p><p>A strength of the work is that the authors train their model on an observable that directly relates to protein dynamics. They also analyse a relatively broad set of proteins which means that one can see actual variation in accuracy across the proteins.</p><p>A weakness of the work is that it is not always clear what the measured R2 rates mean. In some cases, these may include both fast and slow motions (intrinsic R2 rates and exchange contributions). This in turn means that it is actually not clear what the authors are predicting. The work would also be strengthened by making the code available (in addition to the webservice), and by making it easier to compare the accuracy on the training and testing data.</p></body></sub-article><sub-article article-type="author-comment" id="sa3"><front-stub><article-id pub-id-type="doi">10.7554/eLife.88958.3.sa3</article-id><title-group><article-title>Author response</article-title></title-group><contrib-group><contrib contrib-type="author"><name><surname>Qin</surname><given-names>Sanbo</given-names></name><role specific-use="author">Author</role><aff><institution>University of Illinois Chicago</institution><addr-line><named-content content-type="city">Chicago</named-content></addr-line><country>United States</country></aff></contrib><contrib contrib-type="author"><name><surname>Zhou</surname><given-names>Huan-Xiang</given-names></name><role specific-use="author">Author</role><aff><institution>University of Illinois Chicago</institution><addr-line><named-content content-type="city">Chicago</named-content></addr-line><country>United States</country></aff></contrib></contrib-group></front-stub><body><p>The following is the authors’ response to the original reviews.</p><disp-quote content-type="editor-comment"><p>In this useful study, a solid machine learning approach based on a broad set of systems to predict the R2 relaxation rates of residues in intrinsically disordered proteins (IDPs) is described. The ability to predict the patterns of R2 will be helpful to guide experimental studies of IDPs. A potential weakness is that the predicted R2 values may include both fast and slow motions, thus the predictions provide only limited new physical insights into the nature of the relevant protein dynamics.</p></disp-quote><p>Fast motions are less sequence-dependent (e.g., as shown by R1). Hence the sequence-dependent part of R2 singles out slow motion.</p><disp-quote content-type="editor-comment"><p><bold>Public Reviews:</bold></p><p><bold>Reviewer #1 (Public Review):</bold></p><p>Solution state 15N backbone NMR relaxation from proteins reports on the reorientational properties of the N-H bonds distributed throughout the peptide chain. This information is crucial to understanding the motions of intrinsically disordered proteins and as such has focussed the attention of many researchers over the last 20-30 years, both experimentally, analytically and using numerical simulation.</p><p>This manuscript proposes an empirical approach to the prediction of transverse 15N relaxation rates, using a simple formula that is parameterised against a set of 45 proteins. Relaxation rates measured under a wide range of experimental conditions are combined to optimize residuespecific parameters such that they reproduce the overall shape of the relaxation profile. The purely empirical study essentially ignores NMR relaxation theory, which is unfortunate, because it is likely that more insight could have been derived if theoretical aspects had been considered at any level of detail.</p></disp-quote><p>NMR relaxation theory is very valuable in particular regarding motions on different timescales. However, it has very little to say about the sequence dependence of slow motions, which is the focus of our work.</p><disp-quote content-type="editor-comment"><p>Despite some novel aspects, in particular the diversity of the relaxation data sets, the residuespecific parameters do not provide much new insight beyond earlier work that has also noted that sidechain bulkiness correlated with the profile of R2 in disordered proteins.</p></disp-quote><p>The novel insight from our work is that R2 can mostly be predicted based on the local sequence.</p><disp-quote content-type="editor-comment"><p>Nevertheless, the manuscript provides an interesting statistical analysis of a diverse set of deposited transverse relaxation rates that could be useful to the community.</p></disp-quote><p>Thank you!</p><disp-quote content-type="editor-comment"><p>Crucially, and somewhat in contradiction to the authors stated aims in the introduction, I do not feel that the article delivers real insight into the nature of IDP dynamics. Related to this, I have difficulty understanding how an approximate prediction of the overall trend of expected transverse relaxation rates will be of further use to scientists working on IDPs. We already know where the secondary structural elements are (from 13C chemical shifts which are essential for backbone assignment) and the necessary 'scaling' of the profile to match experimental data actually contains a lot of the information that researchers seek.</p></disp-quote><p>Again, the novel insight is that slow motions that dictate the sequence dependence of R2 can mostly be predicted based on the local sequence. The scaling factor may contain useful information but does not tell us anything about the sequence dependence of IDP dynamics.</p><p>This reviewer brings up a lot of valuable points, clearly from an NMR spectroscopist’s perspective. The emphasis of our paper is somewhat different from that perspective. For example, we were interested in whether tertiary contacts make significant contributions to R2, as sometimes claimed. Our results show that, in general, they do not; instead local contacts dominate the sequence dependence of R2.</p><disp-quote content-type="editor-comment"><p>(1) The introduction is confusing, mixing different contributions to R2 as if they emanated from the same physics, which is not necessarily true. 15N transverse relaxation is said to report on 'slower' dynamics from 10s of nanoseconds up to 1 microsecond. Semi-classical Redfield theory shows that transverse relaxation is sensitive to both adiabatic and non-adiabatic terms, due to spin state transitions induced by stochastic motions, and dephasing of coherence due to local field changes, again induced by stochastic motions. These are faster than the relaxation limit dictated by the angular correlation function. Beyond this, exchange effects can also contribute to measured R2. The extent and timescale limit of this contribution depends on the particular pulse sequence used to measure the relaxation. The differences in the pulse sequences used could be presented, and the implications of these differences for the accuracy of the predictive algorithm discussed.</p></disp-quote><p>Indeed pulse sequences affect the measured R2 values. We make the modest assumption that such experimental idiosyncrasy would not corrupt the sequence dependence of IDP dynamics. As for exchange effects, our expectation is that the current SeqDYN may not do well for R2s where slow exchange plays a dominant role in generating sequence dependence, as tertiary contacts would be prominent in those cases; we now present one such case (new Fig. S5).</p><disp-quote content-type="editor-comment"><p>(2) Previous authors have noted the correlation between observed transverse relaxation rates and amino acid sidechain bulkiness. Apart from repeating this observation and optimizing an apparently bulkiness-related parameter on the basis of R2 profiles, I am not clear what more we learn, or what can be derived from such an analysis. If one can possibly identify a motif of secondary structure because raised R2 values in a helix, for example, are missed from the prediction, surely the authors would know about the helix anyway, because they will have assigned the 13C backbone resonances, from which helical propensity can be readily calculated.</p></disp-quote><p>We think that a sequence-based method that is demonstrated to predict well R2 values from expensive NMR experiments is significant. That pi-pi and cation-pi interactions are prominent features of local contacts and may seed tertiary contacts and mediate inter-chain contacts that drive phase separation is a valuable insight.</p><disp-quote content-type="editor-comment"><p>(3) Transverse relaxation rates in IDPs are often measured to a precision of 0.1s-1 or less. This level of precision is achieved because the line-shapes of the resonances are very narrow and high resolution and sensitivity are commonly measurable. The predictions of relaxation rates, even when applying uniform scaling to optimize best-agreement, is often different to experimental measurement by 10 or 20 times the measured accuracy. There are no experimental errors in the figures. These are essential and should be shown for ease of comparison between experiment and prediction.</p></disp-quote><p>Again, our focus is not the precision of the absolute R2 values, but rather the sequence dependence of R2.</p><disp-quote content-type="editor-comment"><p>(4) The impact of structured elements on the dynamic properties of IDPs tethered to them is very well studied in the literature. Slower motions are also increased when, for example the unfolded domain binds a partner, because of the increased slow correlation time. The ad hoc 'helical boosting' proposed by the authors seems to have the opposite effect. When the helical rates are higher, the other rates are significantly reduced. I guess that this is simply a scaling problem. This highlights the limitation of scaling the rates in the secondary structural element by the same value as the rest of the protein, because the timescales of the motion are very different in these regions. In fact the scaling applied by the authors contains very important information. It is also not correct to compare the RMSD of the proposed method with MD, when MD has not applied a 'scaling'. This scaling contains all the information about relative importance of different components to the motion and their timescales, and here it is simply applied and not further analysed.</p></disp-quote><p>Actually, applying the boost factor achieves the effect of a different scaling factor for the secondary structure element than for the rest of the protein.</p><p>Regarding comparing RMSEs of SeqDYN and MD, it is true that SeqDYN applies a scaling factor whereas MD does not. However, even if we apply scaling to MD results it will not change the basic conclusion that “SeqDYN is very competitive against MD in predicting _R_2, but without the significant computational cost.”</p><disp-quote content-type="editor-comment"><p>(5) Generally, the uniform scaling of all values by the same number is serious oversimplification. Motions are happening on all timescales they are giving rise to different transverse relaxation. It is not possible to describe IDP relaxation in terms of one single motion. Detailed studies over more than 30 years, have demonstrated that more than one component to the autocorrelation function is essential in order to account for motions on different timescales in denatured, partially disordered or intrinsically unfolded states. If one could 'scale' everything by the same number, this would imply that only one timescale of motion were important and that all others could be neglected, and this at every site in the protein. This is not expected to be the case, and in fact in the examples shown by the authors it is also never the case. There are always regions where the predicted rates are very different from experiment (with respect to experimental error), presumably because local dynamics are occurring on different timescales to the majority of the molecule. These observations contain useful information, and the observation that a single scaling works quite well probably tells us that one component of the motion is dominant, but not universally. This could be discussed.</p></disp-quote><p>The reviewer appears to equate a single scaling factor with a single type of motion -- this is not correct. A single scaling factor just means that we factor out effects (e.g., temperature or magnetic field) that are uniform across the IDP sequence.</p><disp-quote content-type="editor-comment"><p>(6) With respect to the accuracy of the prediction, discussion about molecular detail such as pi-pi interactions and phase separation propensity is possibly a little speculative.</p></disp-quote><p>It is speculative; we now add more support to this speculation (p. 18 and new Fig. S6).</p><disp-quote content-type="editor-comment"><p>(7) The authors often declare that the prediction reproduces the experimental data. The comparisons with experimental data need to be presented in terms of the chi2 per residue, using the experimentally measured precision which as mentioned, is often very high.</p></disp-quote><p>Again, our interest is the sequence dependence of R2, not the absolute R2 value and its measurement precision.</p><disp-quote content-type="editor-comment"><p><bold>Reviewer #2 (Public Review):</bold></p><p>Qin, Sanbo and Zhou, Huan-Xiang created a model, SeqDYN, to predict nuclear magnetic resonance (NMR) spin relaxation spectra of intrinsically disordered proteins (IDPs), based primarily on amino acid sequence. To fit NMR data, SeqDYN uses 21 parameters, 20 that correspond to each amino acid, and a sequence correlation length for interactions. The model demonstrates that local sequence features impact the dynamics of the IDP, as SeqDYN performs better than a one residue predictor, despite having similar numbers of parameters. SeqDYN is trained using 45 IDP sequences and is retrained using both leave-one-out cross validation and five-fold cross validation, ensuring the model's robustness. While SeqDYN can provide reasonably accurate predictions in many cases, the authors note that improvements can be made by incorporating secondary structure predictions, especially for alpha-helices that exceed the correlation length of the model. The authors apply SeqDYN to study nine IDPs and a denatured ordered protein, demonstrating its predictive power. The model can be easily accessed via the website mentioned in the text.</p><p>While the conclusions of the paper are primarily supported by the data, there are some points that could be extended or clarified.</p><p>(1) The authors state that the model includes 21 parameters. However, they exclude a free parameter that acts as a scaling factor and is necessary to fit the experimental data (lambda). As a result, SeqDYN does not predict the spectrum from the sequence de-novo, but requires a one parameter fitting. The authors mention that this factor is necessary due to non-sequence dependent factors such as the temperature and magnetic field strength used in the experiment.</p><p>Given these considerations, would it be possible to predict what this scaling factor should be based on such factors?</p></disp-quote><p>There are still too few data to make such a prediction.</p><disp-quote content-type="editor-comment"><p>(2) The authors mention that the Lorentzian functional form fits the data better than a Gaussian functional form, but do not present these results.</p></disp-quote><p>We tested the different functional forms at the early stage of the method development. The improvement of the Lorentzian over the Gaussian was slight and we simply decided on the Lorentzian and did not go back and do a systematic analysis.</p><disp-quote content-type="editor-comment"><p>(3) The authors mention that they conducted five-fold cross validation to determine if differences between amino acid parameters are statistically significant. While two pairs are mentioned in the text, there are 190 possible pairs, and it would be informative to more rigorously examine the differences between all such pairs.</p></disp-quote><p>We now present t-test results for other pairs in new Fig. S3.</p><disp-quote content-type="editor-comment"><p><bold>Reviewer #3 (Public Review):</bold></p><p>The manuscript by Qin and Zhou presents an approach to predict dynamical properties of an intrinsically disordered protein (IDP) from sequence alone. In particular, the authors train a simple (but useful) machine learning model to predict (rescaled) NMR R2 values from sequence. Although these R2 rates only probe some aspects of IDR dynamics and the method does not provide insight into the molecular aspects of processes that lead to perturbed dynamics, the method can be useful to guide experiments.</p><p>A strength of the work is that the authors train their model on an observable that directly relates to protein dynamics. They also analyse a relatively broad set of proteins which means that one can see actual variation in accuracy across the proteins.</p><p>A weakness of the work is that it is not always clear what the measured R2 rates mean. In some cases, these may include both fast and slow motions (intrinsic R2 rates and exchange contributions). This in turn means that it is actually not clear what the authors are predicting. The work would also be strengthened by making the code available (in addition to the webservice), and by making it easier to compare the accuracy on the training and testing data.</p></disp-quote><p>Our method predicts the sequence dependence of R2, which is dominated by slower dynamics.</p><disp-quote content-type="editor-comment"><p><bold>Recommendations for the authors:</bold></p><p><bold>Reviewer #2 (Recommendations For The Authors):</bold></p><p>(1) Should make sure to define abbreviations such as NMR and SeqDYN.</p></disp-quote><p>We now spell out NMR at first use. SeqDYN is the name of our method and is not an abbreviation.</p><disp-quote content-type="editor-comment"><p>(2) The authors do not mention how the curves in Figure 2A are calculated.</p></disp-quote><p>As we stated in the figure caption, these curves are drawn to guide the eye.</p><disp-quote content-type="editor-comment"><p>(3) May be interesting to explore how the model parameters (q) correlate with different measures of hydrophobicity (especially those derived for IDPs like Urry). This may point to a relationship between amino acid interactions and amino acid dynamics</p></disp-quote><p>We now present the correlation between q and a stickiness parameter refined by Tesei et al. (new ref 45) and used for predicting phase separation equilibrium (new Fig. S6).</p><disp-quote content-type="editor-comment"><p>(4) The authors demonstrate that secondary structure cannot be fully accounted for by their model. They make a correction for extended alpha-helices, but the strength of this correction seems to only be based on one sequence. Would a more rigorous secondary structure correction further improve the model and perhaps allow its transferability to ordered proteins?</p></disp-quote><p>We have five 4 test cases (Figs. 4E, F and 5H, I). However, we doubt that the SeqDYN method will be transferable to ordered proteins.</p><disp-quote content-type="editor-comment"><p><bold>Reviewer #3 (Recommendations For The Authors):</bold></p><p>Changes that could strengthen the manuscript substantially.</p><p>(1) The authors do not really define what they mean by dynamics, but given that they train and benchmark on R2 measurements, the directly probe whatever goes into the measured R2. Using a direct measurement is a strength since it makes it clear what they are predicting. It also, however, makes it difficult to interpret. This is made clear in the text when the authors, for example write &quot;𝑅2 is the one most affected by slower dynamics (10s of ns to 1 μs and beyond).&quot; First, with the &quot;and beyond&quot; it could literally mean anything. Second, the &quot;normal&quot; R2 rate is limited up to motions up to the (local) &quot;tumbling/reorganization&quot; time (which is much faster), so any slow motions that go into R2 would be what one would normally call &quot;exchange&quot;. The authors should thus make it clearer what exactly it is they are probing. In the end, this also depends on the origin of the experimental data, and whether the &quot;R2&quot; measurements are exchange-free or not. This may be a mixture, which hampers interpretations and which may also explain some of the rescaling that needs to be done.</p></disp-quote><p>We now remove “and beyond”, and also raise the possibility that R2 measurements based on 15N relaxation may have relatively small exchange contributions (p. 17).</p><disp-quote content-type="editor-comment"><p>(2) Related to the above, the authors might consider comparing their predictions to the relaxation experiments from Kriwacki and colleagues on a fragment of p27. In that work, the authors used dispersion experiments to probe the dynamics on different timescales. The authors would here be able to compare both to the intrinsic R2 rates (when slow motions are pulsed away) as well as the effective R2 rates (which would be the most common measurement). This would help shed light on (at least in one case) which type of R2 the prediction model captures. <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1021/jacs.7b01380">https://doi.org/10.1021/jacs.7b01380</ext-link></p></disp-quote><p>We now report this comparison in new Fig. S5 and discuss its implications (p. 17-18).</p><disp-quote content-type="editor-comment"><p>(3) In some cases, disagreement between prediction and experiments is suggested to be due to differences in temperature, and hence is used as an argument for the rescaling done. Here, the authors use a factor of 2.0 to explain a difference between 278K and 298K, and a factor of 2.4 to explain the difference between 288K and 298K. It would be surprising if the temperature effect from 288K-&gt;298K is larger than from 278K-&gt;298K. Does this not suggest that the differences come as much from other sources?</p></disp-quote><p>Note that the scaling factors 2.0 and 2.4 were obtained on two different IDPs. It is most likely that different IDPs have different scaling factors for temperature change. As a simple model, the tumbling time for a spherical particle scales with viscosity and the particle volume; correspondingly the scaling factor for temperature change should be greater for a larger particle than for a smaller particle.</p><disp-quote content-type="editor-comment"><p>(4) The authors find (as have others before) aromatic residues to be common at/near R2 peaks. They suggest this to be indicative for Pi-Pi interactions. Could this not be other types of interactions since these residues are also &quot;just&quot; more hydrophobic? Also, can the authors rule out that the increased R2 rates near aromatic residues is not due to increased dynamics, but simply due to increased Rex-terms due to greater fluctuations in the chemical shifts near these residues (due to the large ring current effects).</p></disp-quote><p>We noted both pi-pi and cation-pi as possible interactions that raise R2. There can be other interactions involving aromatic residues, but it’s unlikely to be only hydrophobic as Arg is also in the high-<italic>q</italic> end. For the same reason, a ring-current based explanation would be inadequate.</p><disp-quote content-type="editor-comment"><p>(5) The authors write: &quot;We found that, by filtering PsiPred (<ext-link ext-link-type="uri" xlink:href="http://bioinf.cs.ucl.ac.uk/psipred">http://bioinf.cs.ucl.ac.uk/psipred</ext-link>) (35) helix propensity scores (𝑝,-.) with a very high cutoff of 0.99, the surviving helix predictions usually correspond well with residues identified by NMR as having high helix propensities.&quot; It would be good to show the evidence for this in the paper, and quantify this statement.</p></disp-quote><p>The cases of most interest are the ones with long predicted helices, of which there are only 3 in the training set. For Sev-NT and CBP-ID4, we already summarize the NMR data for helix identification in the first paragraph of Results; the third case is KRS-NT, which we elaborate in p. 14.</p><disp-quote content-type="editor-comment"><p>(6) When analysing the nine test proteins, it would be very useful for the reader to get a number for the average accuracy on the nine proteins and a corresponding number for the training proteins. The numbers are maybe there, but hard to find/compare. This would be important so that one can understand how well the model works on the training vs testing data.</p></disp-quote><p>We now present the mean RMSE comparison in p. 14.</p><disp-quote content-type="editor-comment"><p>(7) The authors write: &quot;The 𝑞 parameters, while introduced here to characterize the propensities of amino acids to participate in local interactions, appear to correlate with the tendencies of amino acids to drive liquid-liquid phase separation.&quot; It would be good to show this data and quantify this.</p></disp-quote><p>We now list supporting data in p. 18 and present new Fig. S6 for further support.</p><disp-quote content-type="editor-comment"><p>(8) It is great that the authors have made a webservice available for easy access to the work. They should in my opinion also make the training code and data available, as well as the final trained model. Here it would also be useful to show the results from the use of a Gaussian that was also tested, and also state whether this model was discarded before or after examining the testing data.</p></disp-quote><p>We have listed the IDP characteristics and sequences in Tables S1 and S2. We’re unsure whether we can disseminate the experimental R2 data without the permission of the original authors. As for the Gaussian function, as stated above, it was abandoned at an early state, before examining the testing data.</p><disp-quote content-type="editor-comment"><p>Changes that would also be useful</p><p>(1) The authors should make it clearer what they predict and what they don't. They mention transient helix formation and various contacts, but there isn't a one-to-one relationship between these structural features and R2 rates. Hence, they should make it clearer that they don't predict secondary structure and that an increased R2 rate may be indicative of many different structural/dynamical features on many different time scales.</p></disp-quote><p>We clearly state that we apply a helix boost after the regular SeqDYN prediction.</p><disp-quote content-type="editor-comment"><p>(2) The authors write &quot;Instead, dynamics has emerged as a crucial link between sequence and function for IDPs&quot; and cite their own work (reference 1) as reference for this statement. As far as I can see, that work does not study function of IDPs. Maybe the authors could cite additional work showing that the dynamics (time scales) affects function of IDPs beyond &quot;just&quot; structure? Otherwise, the functional consequences are not clear. Maybe the authors mean that R2 rates are indicative of (residual) structure, but that is not quite the same. Also, even in that case, there are likely more appropriate references.</p></disp-quote><p>Ref. 1 summarized a number of scenarios where dynamics is related to function.</p><disp-quote content-type="editor-comment"><p>(3) The authors might want to look at some of the older literature on interpreting NMR relaxation rates and consider whether some of it is worth citing.</p><p>Fitting/understanding R2 profiles <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1021/bi020381o">https://doi.org/10.1021/bi020381o</ext-link> <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1007/s10858-006-9026-9">https://doi.org/10.1007/s10858-006-9026-9</ext-link></p><p>MD simulations and comparisons to R2 rates without ad hoc reweighting (in addition to the papers from the authors themselves). <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1021/ja710366c">https://doi.org/10.1021/ja710366c</ext-link> <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1021/ja209931w">https://doi.org/10.1021/ja209931w</ext-link></p></disp-quote><p>The R2 data for the two unfolded proteins are very helpful! We now present the comparison of these data to SeqDYN prediction in Fig. 6C, D. The MD papers are superseded by more recent studies (e.g., refs. 1 and 14).</p><disp-quote content-type="editor-comment"><p>There are more like these.</p><p>(4) In the analysis of unfolded lysozyme, I assume that the authors are treating the methylated cysteines (which are used in the experiments) simply as cysteine. If that is the case, the authors should ideally mention this specifically.</p></disp-quote><p>Treatment of methylated cysteines is now stated in the Fig. 6 caption.</p><disp-quote content-type="editor-comment"><p>(5) The authors write &quot;Pro has an excessively low ms𝑅2 [with data from only two IDPs (32, 33)], but that is due to the absence of an amide proton.&quot; It would be useful with an explanation why lacking a proton gives rise to low 15N R2 rates.</p></disp-quote><p>That assertion originated from ref. 32.</p><disp-quote content-type="editor-comment"><p>(6) When applying the model, the authors predict msR2 and then compare to experimental R2 by rescaling with a factor gamma. It would be good to make it clearer whether this parameter is always fitted to the experiments in all the comparisons. It would be useful to list the fitted gamma values for all the proteins (e.g. in Table S1).</p></disp-quote><p>We already give a summary of the scaling factors (“For 39 of the 45 IDPs, Υ values fall in the range of 0.8 to 2.0 s–1”, p. 10).</p><disp-quote content-type="editor-comment"><p>(7) p. 14 &quot;nineth&quot; -&gt; &quot;ninth&quot;</p></disp-quote><p>Corrected</p></body></sub-article></article>