<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Archiving and Interchange DTD with MathML3 v1.2 20190208//EN"  "JATS-archivearticle1-mathml3.dtd"><article xmlns:ali="http://www.niso.org/schemas/ali/1.0/" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article" dtd-version="1.2"><front><journal-meta><journal-id journal-id-type="nlm-ta">elife</journal-id><journal-id journal-id-type="publisher-id">eLife</journal-id><journal-title-group><journal-title>eLife</journal-title></journal-title-group><issn publication-format="electronic" pub-type="epub">2050-084X</issn><publisher><publisher-name>eLife Sciences Publications, Ltd</publisher-name></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">84412</article-id><article-id pub-id-type="doi">10.7554/eLife.84412</article-id><article-categories><subj-group subj-group-type="display-channel"><subject>Research Article</subject></subj-group><subj-group subj-group-type="heading"><subject>Structural Biology and Molecular Biophysics</subject></subj-group></article-categories><title-group><article-title>Peptides that Mimic RS repeats modulate phase separation of SRSF1, revealing a reliance on combined stacking and electrostatic interactions</article-title></title-group><contrib-group><contrib contrib-type="author" id="author-297078"><name><surname>Fargason</surname><given-names>Talia</given-names></name><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0001-6888-0356</contrib-id><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="fn" rid="con1"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author" id="author-297079"><name><surname>De Silva</surname><given-names>Naiduwadura Ivon Upekala</given-names></name><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0002-5937-0271</contrib-id><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="fn" rid="con2"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author" id="author-297080"><name><surname>Powell</surname><given-names>Erin</given-names></name><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="fn" rid="con3"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author" id="author-297081"><name><surname>Zhang</surname><given-names>Zihan</given-names></name><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="fn" rid="con4"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author" id="author-309477"><name><surname>Paul</surname><given-names>Trenton</given-names></name><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0009-0000-0931-3888</contrib-id><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="fn" rid="con5"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author" id="author-297083"><name><surname>Shariq</surname><given-names>Jamal</given-names></name><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="fn" rid="con6"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author" id="author-297084"><name><surname>Zaharias</surname><given-names>Steve</given-names></name><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="fn" rid="con7"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author" corresp="yes" id="author-296648"><name><surname>Zhang</surname><given-names>Jun</given-names></name><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0002-5842-7424</contrib-id><email>zhanguab@uab.edu</email><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="other" rid="fund1"/><xref ref-type="other" rid="fund2"/><xref ref-type="fn" rid="con8"/><xref ref-type="fn" rid="conf1"/></contrib><aff id="aff1"><label>1</label><institution-wrap><institution-id institution-id-type="ror">https://ror.org/008s83205</institution-id><institution>Department of Chemistry, University of Alabama at Birmingham</institution></institution-wrap><addr-line><named-content content-type="city">Birmingham</named-content></addr-line><country>United States</country></aff></contrib-group><contrib-group content-type="section"><contrib contrib-type="editor"><name><surname>Black</surname><given-names>Douglas L</given-names></name><role>Reviewing Editor</role><aff><institution-wrap><institution-id institution-id-type="ror">https://ror.org/046rm7j60</institution-id><institution>University of California, Los Angeles</institution></institution-wrap><country>United States</country></aff></contrib><contrib contrib-type="senior_editor"><name><surname>Manley</surname><given-names>James L</given-names></name><role>Senior Editor</role><aff><institution-wrap><institution-id institution-id-type="ror">https://ror.org/00hj8s172</institution-id><institution>Columbia University</institution></institution-wrap><country>United States</country></aff></contrib></contrib-group><pub-date publication-format="electronic" date-type="publication"><day>02</day><month>03</month><year>2023</year></pub-date><pub-date pub-type="collection"><year>2023</year></pub-date><volume>12</volume><elocation-id>e84412</elocation-id><history><date date-type="received" iso-8601-date="2022-10-24"><day>24</day><month>10</month><year>2022</year></date><date date-type="accepted" iso-8601-date="2023-03-01"><day>01</day><month>03</month><year>2023</year></date></history><pub-history><event><event-desc>This manuscript was published as a preprint at .</event-desc><date date-type="preprint" iso-8601-date="2022-10-24"><day>24</day><month>10</month><year>2022</year></date><self-uri content-type="preprint" xlink:href="https://doi.org/10.1101/2022.10.24.511151"/></event></pub-history><permissions><copyright-statement>© 2023, Fargason et al</copyright-statement><copyright-year>2023</copyright-year><copyright-holder>Fargason et al</copyright-holder><ali:free_to_read/><license xlink:href="http://creativecommons.org/licenses/by/4.0/"><ali:license_ref>http://creativecommons.org/licenses/by/4.0/</ali:license_ref><license-p>This article is distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="http://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution License</ext-link>, which permits unrestricted use and redistribution provided that the original author and source are credited.</license-p></license></permissions><self-uri content-type="pdf" xlink:href="elife-84412-v2.pdf"/><self-uri content-type="figures-pdf" xlink:href="elife-84412-figures-v2.pdf"/><abstract><p>Phase separation plays crucial roles in both sustaining cellular function and perpetuating disease states. Despite extensive studies, our understanding of this process is hindered by low solubility of phase-separating proteins. One example of this is found in SR and SR-related proteins. These proteins are characterized by domains rich in arginine and serine (RS domains), which are essential to alternative splicing and in vivo phase separation. However, they are also responsible for a low solubility that has made these proteins difficult to study for decades. Here, we solubilize the founding member of the SR family, SRSF1, by introducing a peptide mimicking RS repeats as a co-solute. We find that this RS-mimic peptide forms interactions similar to those of the protein’s RS domain. Both interact with a combination of surface-exposed aromatic residues and acidic residues on SRSF1’s RNA Recognition Motifs (RRMs) through electrostatic and cation-pi interactions. Analysis of RRM domains from human SR proteins indicates that these sites are conserved across the protein family. In addition to opening an avenue to previously unavailable proteins, our work provides insight into how SR proteins phase separate and participate in nuclear speckles.</p></abstract><kwd-group kwd-group-type="author-keywords"><kwd>SRSF1</kwd><kwd>phase separation</kwd><kwd>cation-pi interaction</kwd><kwd>intrinsically disordered protein</kwd><kwd>NMR</kwd><kwd>SR proteins</kwd></kwd-group><kwd-group kwd-group-type="research-organism"><title>Research organism</title><kwd>Human</kwd></kwd-group><funding-group><award-group id="fund1"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100000001</institution-id><institution>National Science Foundation</institution></institution-wrap></funding-source><award-id>MCB2024964</award-id><principal-award-recipient><name><surname>Zhang</surname><given-names>Jun</given-names></name></principal-award-recipient></award-group><award-group id="fund2"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100000002</institution-id><institution>National Institutes of Health</institution></institution-wrap></funding-source><award-id>R35GM147091</award-id><principal-award-recipient><name><surname>Zhang</surname><given-names>Jun</given-names></name></principal-award-recipient></award-group><funding-statement>The funders had no role in study design, data collection and interpretation, or the decision to submit the work for publication.</funding-statement></funding-group><custom-meta-group><custom-meta specific-use="meta-only"><meta-name>Author impact statement</meta-name><meta-value>Short peptides that mimic the regions responsible for phase separation can be used to solubilize phase-separating proteins in their native states and to determine the mechanism of phase separation.</meta-value></custom-meta></custom-meta-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>Liquid-liquid phase separation underpins the formation of membraneless organelles, such as nucleoli (<xref ref-type="bibr" rid="bib43">Lafontaine et al., 2021</xref>), P-bodies (<xref ref-type="bibr" rid="bib9">Brangwynne et al., 2009</xref>), stress granules (<xref ref-type="bibr" rid="bib52">Molliex et al., 2015</xref>), cajal bodies (<xref ref-type="bibr" rid="bib56">Neugebauer, 2017</xref>), and nuclear speckles (<xref ref-type="bibr" rid="bib23">Fei et al., 2017</xref>). The integrity of such organelles is maintained by interactions between biomolecules that form condensates, or liquid droplet-like structures, in which the local concentration of individual components is higher than the surrounding environment (<xref ref-type="bibr" rid="bib78">Yang et al., 2004</xref>). These condensates cluster relevant molecules together to facilitate interactions while allowing rapid material exchange (<xref ref-type="bibr" rid="bib78">Yang et al., 2004</xref>; <xref ref-type="bibr" rid="bib67">Souquere et al., 2009</xref>; <xref ref-type="bibr" rid="bib19">Dundr and Misteli, 2010</xref>; <xref ref-type="bibr" rid="bib32">Handwerger et al., 2005</xref>). Mounting evidence has revealed roles of phase separation in modulating reaction kinetics, enzyme catalysis, and binding specificity (<xref ref-type="bibr" rid="bib61">Reber et al., 2021</xref>; <xref ref-type="bibr" rid="bib5">Banani et al., 2017</xref>; <xref ref-type="bibr" rid="bib68">Strulson et al., 2012</xref>; <xref ref-type="bibr" rid="bib6">Banjade and Rosen, 2014</xref>; <xref ref-type="bibr" rid="bib47">Li et al., 2012</xref>).</p><p>The protein SRSF1 (Serine/Arginine-Rich Splicing Factor 1, also known as ASF/SF2) is essential for the early-stage assembly of the spliceosome (<xref ref-type="bibr" rid="bib40">Kohtz et al., 1994</xref>; <xref ref-type="bibr" rid="bib14">Cho et al., 2011</xref>). Several in vivo studies have shown that SRSF1 is found in condensates (<xref ref-type="bibr" rid="bib23">Fei et al., 2017</xref>; <xref ref-type="bibr" rid="bib31">Hammarskjold and Rekosh, 2017</xref>; <xref ref-type="bibr" rid="bib33">Haward et al., 2021</xref>; <xref ref-type="bibr" rid="bib45">Lamond and Spector, 2003</xref>; <xref ref-type="bibr" rid="bib4">Azpurua et al., 2021</xref>; <xref ref-type="bibr" rid="bib34">Ilik et al., 2020</xref>; <xref ref-type="bibr" rid="bib49">Li and Wang, 2021</xref>). Aberrant condensation behaviors have also been observed in disease states (<xref ref-type="bibr" rid="bib4">Azpurua et al., 2021</xref>; <xref ref-type="bibr" rid="bib34">Ilik et al., 2020</xref>; <xref ref-type="bibr" rid="bib49">Li and Wang, 2021</xref>). SRSF1 belongs to the Ser/Arg-rich protein family (SR proteins), which contains 12 members possessing one to two structured RNA-recognition motifs (RRMs) and a repetitive Arg/Ser repeat region (RS domain; <xref ref-type="bibr" rid="bib66">Shepard and Hertel, 2009</xref>; <xref ref-type="bibr" rid="bib69">Tacke and Manley, 1995</xref>; <xref ref-type="bibr" rid="bib65">Screaton et al., 1995</xref>). The RS region is also found in the much larger family of SR-related proteins, which contain the repetitive RS regions but not the other structural features (<xref ref-type="bibr" rid="bib7">Blencowe et al., 1999</xref>; <xref ref-type="bibr" rid="bib13">Cascarina and Ross, 2022</xref>). Two SR-related proteins, SRRM2 (serine-arginine rich repetitive matrix protein 2) and SON, are essential for the formation and structural maintenance of the membraneless organelles nuclear speckles (<xref ref-type="bibr" rid="bib1">Ahn et al., 2011</xref>; <xref ref-type="bibr" rid="bib77">Xu et al., 2022</xref>). Like many other splicing factors, SRSF1 modulates trafficking to the speckles (<xref ref-type="bibr" rid="bib71">Tripathi et al., 2012</xref>). It has been demonstrated that SRSF1 is found in nuclear speckles when its RS domain is partially phosphorylated but that hyperphosphorylation causes the protein to leave nuclear speckles (<xref ref-type="bibr" rid="bib3">Aubol et al., 2018</xref>; <xref ref-type="bibr" rid="bib29">Gui et al., 1994</xref>). It is therefore evident that RS domains play an important role in the organization of nuclear speckles. However, an understanding of the nature of that interaction has been evasive due to a difficulty solubilizing the proteins involved. As with all 12 SR proteins and many speckle components, obtaining soluble SRSF1 has been an imposing challenge for decades, and this has substantially hindered our understanding of the functions of these proteins and of nuclear speckles as a whole (<xref ref-type="bibr" rid="bib66">Shepard and Hertel, 2009</xref>).</p><p>Phase separation is frequently mediated by repetitive sequences. Here, we find this to be the case for the protein SRSF1, whose phase separation is dependent on its RS repeats. Our bioinformatic analysis reveals a correlation of RS repeats with a tendency to phase separate. We successfully solubilize SRSF1 using short peptides that mimic RS repeats. Our success in solubilizing SRSF1 provides an unprecedented opportunity to elucidate the mechanism by which RS repeats interact with SRSF1. We find that this increase in solubility is due to a competition between the peptide and RS domains for the same binding sites on RRM domains. We further discover that acidic residues and aromatic residues from SRSF1 RRMs interact with the RS region through salt bridges and cation-pi interactions. We find that many of the RRM sites interacting with RS repeats are conserved among the SR protein family. These findings provide insight into how interactions between SR and SR-related proteins may occur within membraneless organelles. They also allow us to predict how the nature of these interactions might change when the RS domain becomes phosphorylated.</p></sec><sec id="s2" sec-type="results"><title>Results</title><sec id="s2-1"><title>RS repeats are abundant in the human proteome and associated with phase separation</title><p>RS repeats are often found in SR proteins and SR-related proteins. The serine residues in RS repeats are frequently phosphorylated, a process which regulates the functions of RS repeats. To quantify the abundance of RS repeats in the human proteome, we systematically searched for uninterrupted repeats that were 2–8 amino acids in length (<xref ref-type="fig" rid="fig1">Figure 1A</xref>) and tested whether the length of RS repeats was correlated with protein condensates. Proteins were defined as being in condensates if they were listed in any of three available phase separation databases (PhaSepDB, LLPSDB, and DrLLPS) (<xref ref-type="bibr" rid="bib58">Ning et al., 2020</xref>; <xref ref-type="bibr" rid="bib48">Li et al., 2020</xref>; <xref ref-type="bibr" rid="bib79">You et al., 2020</xref>). Because both RRM domains and RS repeats have been associated with phase separation in previous studies, we separated proteins containing RRM domains in this analysis from those that did not (<xref ref-type="bibr" rid="bib72">Wang et al., 2018</xref>; <xref ref-type="bibr" rid="bib54">Murthy et al., 2019</xref>). We found that the chance that a protein was found in condensates increased with the length of RS repeats it harbored regardless of whether the protein contained an RRM domain (<xref ref-type="fig" rid="fig1">Figure 1A</xref>). In our analysis, we did not distinguish RS repeats from SR repeats, and counting started with the first residue, whether it was R or S. We analyzed the correlation between the RS repeat length and the percentage of proteins found in condensates using correlation analysis (<xref ref-type="supplementary-material" rid="supp1">Supplementary file 1</xref>) and contingency tables (<xref ref-type="supplementary-material" rid="supp3">Supplementary file 3</xref>). Using correlation analysis, we found that the two-tailed Pearson’s <italic>p</italic>-value is 0.02, and the correlation coefficient is 0.93 (<xref ref-type="supplementary-material" rid="supp1">Supplementary file 1</xref>). We further extended the correlation analysis to all other possible dipeptide motifs, assuming that the order of the two amino acids in a repeat is interchangeable. As the protein population of some dipeptide motifs is low, we estimated a population-based error of <inline-formula><mml:math id="inf1"><mml:mrow><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mo>/</mml:mo><mml:mrow><mml:msqrt><mml:msub><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>p</mml:mi><mml:mi>s</mml:mi></mml:mrow></mml:msub></mml:msqrt></mml:mrow></mml:mrow></mml:math></inline-formula> , where <italic>N<sub>ps</sub></italic> is the number of proteins containing 8-mer peptides found in condensates (as described more completely in the methods section). Applying a criterion of p-value &lt;0.05 and fraction of proteins in condensates greater than twice the population-based error, we found six dipeptide motifs showed significant correlation with phase separation: GG, KK, QQ, PP, RG, and RS (<xref ref-type="supplementary-material" rid="supp1">Supplementary file 1</xref>). Except for the KK motif, the five other dipeptide repeats have been shown to directly drive phase separation for some proteins (<xref ref-type="bibr" rid="bib43">Lafontaine et al., 2021</xref>; <xref ref-type="bibr" rid="bib9">Brangwynne et al., 2009</xref>; <xref ref-type="bibr" rid="bib52">Molliex et al., 2015</xref>; <xref ref-type="bibr" rid="bib23">Fei et al., 2017</xref>). It is noteworthy that among the six dipeptide motifs, RS-containing proteins have the highest percentage of 6-mer and 8-mer-containing proteins in condensates. As the sample sizes across these datasets vary widely, we performed the same analysis on 50 randomly selected size-matched subsets from each category (<xref ref-type="supplementary-material" rid="supp2">Supplementary file 2</xref>). We obtained similar results when size-matched datasets were used.</p><fig-group><fig id="fig1" position="float"><label>Figure 1.</label><caption><title>A combination of RS repeats and RRM domains is highly correlated with appearance in condensates.</title><p>(<bold>A</bold>) Increased RS repeat length leads to an increased likelihood of appearance in condensates. Percentage of proteins possessing indicated properties that appear in one of three major phase separation databases. The Pearson’s p-value (0.02) for the correlation between RS length and phase separation likelihood is shown in <xref ref-type="supplementary-material" rid="supp1">Supplementary file 1</xref>. Correlation between RS and RRM occurrence was analyzed by Fisher’s exact test (<xref ref-type="supplementary-material" rid="supp3">Supplementary file 3</xref>). (<bold>B</bold>) Correlation between number of 2-mer RS and 4-mer RS repeats with appearance in condensates. Proteins found in condensates are more likely to have a greater number of RS dipeptide and tetra-peptide repeats in the absence of RRM domains (-). The p-values presented were obtained using the Mann-Whitney test, which is suitable for non-normal distributions with different sample sizes (<xref ref-type="bibr" rid="bib73">Widen et al., 2020</xref>). Bonferroni’s adjustment was applied to adjust the significance level to p-value = 0.025. (<bold>C</bold>) Proteins with <underline>&gt;</underline>10 2 mer RS repeats or <underline>&gt;</underline>2 4 mer RS repeats are more likely to phase separate, particularly when RRM domains are present. p<italic>-</italic>values were calculated using Fisher’s exact test (<xref ref-type="supplementary-material" rid="supp4">Supplementary file 4</xref>).</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-84412-fig1-v2.tif"/></fig><fig id="fig1s1" position="float" specific-use="child-fig"><label>Figure 1—figure supplement 1.</label><caption><title>Increased repeat number and R/S percent composition correlate with increased phase separation.</title><p>(<bold>A</bold>) Amino-acid sequence of the protein SRRM2 with 56 4-mer RS repeats highlighted in yellow. Red serines can be phosphorylated according to the Uniprot database. Deletion of the undeca-repeat (UPR) and dodeca-repeat (DPR) regions results in dissociation of nuclear speckles (<xref ref-type="bibr" rid="bib77">Xu et al., 2022</xref>). (<bold>B</bold>) Effect of percent R/S composition on the likelihood of phase separation. Each point indicates the 20-amino acid sequence in the protein that is densest in R/S. Percentages were calculated using LCD-Composer (<xref ref-type="bibr" rid="bib12">Cascarina et al., 2021</xref>). Minimum density was set to 5%. p-values were obtained using the Mann-Whitney test. (<bold>C</bold>) Proteins with a 20-amino acid sequence of ≥40% R/S composition are more likely to be found in condensates, particularly when RRM domains are present. p-values were calculated using Fisher’s exact test.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-84412-fig1-figsupp1-v2.tif"/></fig></fig-group><p>In addition to the effect of increased RS repeat length, we also found that proteins possessing both RS repeats and RRM domains were especially likely to be found in condensates (<xref ref-type="fig" rid="fig1">Figure 1A</xref>, blue bars). Among proteins with an RRM and at least one 4-mer RS repeat, the likelihood of appearance in condensates was 89%, and this trend became more pronounced as the repeat length was increased (<xref ref-type="fig" rid="fig1">Figure 1A</xref>). Using contingency tables, we analyzed the correlation between the presence of RS dipeptides and the occurrence of RRM domains (<xref ref-type="supplementary-material" rid="supp3">Supplementary file 3</xref>). For proteins found in condensates, occurrence of RS dipeptides and RRM domains is clearly correlated, as shown by a p-value of 0.0013 (<xref ref-type="supplementary-material" rid="supp3">Supplementary file 3</xref>). In contrast, no significant correlation was found when the same analysis was performed on proteins not in condensates (p-value = 0.1202). As repeat length increased, the correlation between RS repeats and RRM domains also increased.</p><p>Many proteins contain multiple copies of short RS repeats. In fact, most SR and SR-related proteins have several short RS repeats instead of a few long, continuous ones (<xref ref-type="bibr" rid="bib8">Boucher et al., 2001</xref>). One extreme example is nuclear speckle scaffolding protein SRRM2, which has 56 4-mer RS repeats, more than any other protein (<xref ref-type="fig" rid="fig1s1">Figure 1—figure supplement 1A</xref>). Therefore, we also analyzed how the number of short RS repeats is correlated with phase separation. We found that the number of RS repeats that a protein harbors also affects its likelihood of being found in condensates (<xref ref-type="fig" rid="fig1">Figure 1B and C</xref>). In the absence of an RRM domain, on average, proteins in condensates have more copies of 2-mer or 4-mer RS repeats than those not in condensates (<xref ref-type="fig" rid="fig1">Figure 1B</xref>). Further, proteins with several RS repeats and an RRM domain are particularly likely to be found in condensates (<xref ref-type="fig" rid="fig1">Figure 1C</xref>, <xref ref-type="supplementary-material" rid="supp4">Supplementary file 4</xref>). Definitions of RS domains usually specify either a threshold repeat number (<xref ref-type="bibr" rid="bib8">Boucher et al., 2001</xref>; <xref ref-type="bibr" rid="bib50">Manley and Krainer, 2010</xref>) or a threshold percentage R/S composition (<xref ref-type="bibr" rid="bib13">Cascarina and Ross, 2022</xref>; <xref ref-type="bibr" rid="bib50">Manley and Krainer, 2010</xref>). To test the effect of percent R/S composition on the likelihood of phase separation, we used LCD composer (<xref ref-type="bibr" rid="bib12">Cascarina et al., 2021</xref>). Similar to the results we found for the effect of short repeats, we found that a 20-amino acid sequence of at least 40% RS composition increased the likelihood of a protein being found in condensates from 31% to 36% (<xref ref-type="fig" rid="fig1s1">Figure 1—figure supplement 1B and C</xref>, <xref ref-type="supplementary-material" rid="supp4">Supplementary file 4</xref>). Further addition of at least one RRM domain increased the likelihood of phase separation to 89% (<xref ref-type="fig" rid="fig1s1">Figure 1—figure supplement 1C</xref>). In summary, we found a correlation between RS repeats and phase separation whether it was analyzed by length, number, or composition of RS repeats.</p></sec><sec id="s2-2"><title>SRSF1 can be solubilized using peptides that mimic its RS region</title><p>The correlation of RS-repeats with phase separation is consistent with observations that many RS- containing proteins have low solubility in vitro. For example, up to this point, none of the full-length SR proteins have been obtained in concentrations suitable for biophysical/biochemical or structural characterization, although the founding member of the family, SRSF1, was identified more than three decades ago (<xref ref-type="bibr" rid="bib41">Krainer et al., 1990a</xref>; <xref ref-type="bibr" rid="bib25">Ge and Manley, 1990</xref>; <xref ref-type="bibr" rid="bib42">Krainer et al., 1990b</xref>).</p><p>To overcome this obstacle in investigating SR and SR-related proteins, we aimed to develop a new purification and solubilization method using full-length SRSF1. The high Arg composition in these proteins inspired us to use high concentrations of Arg amino acid in our protocol to purify and solubilize SRSF1. An Arg/Glu mixture of 50 mM has been used to increase solubility of some RNA-binding proteins (<xref ref-type="bibr" rid="bib26">Golovanov et al., 2004</xref>). We found that 0.8–1 M of Arg was able to solubilize all SRSF1 constructs during the purification procedure (details in the Materials and methods section). However, the high ionic strength of Arg at this concentration range is unsuitable for many analytical methods, such as NMR and binding assays.</p><p>We predicted that we could solubilize phase-separating proteins using peptide co-solutes that mimic these repeats to compete with inter- and intra- molecular interactions (<xref ref-type="fig" rid="fig2">Figure 2A</xref>). We tested this concept on SRSF1, which contains RS repeats of 16, 5, and 6 amino acids, respectively (<xref ref-type="fig" rid="fig2">Figure 2B</xref>). Serine residues in these regions can be phosphorylated, resulting in an alternation of positive and negative charges (<xref ref-type="fig" rid="fig2">Figure 2B</xref>). To solubilize SRSF1 in its unphosphorylated and phosphorylated forms, we therefore tested peptides of varying lengths to mimic unphosphorylated RS (RS), and phosphorylated RS (DR, and ER) repeats (<xref ref-type="fig" rid="fig2">Figure 2B</xref>). Here, using the purified protein, we found that SRSF1 phase separated at concentrations lower than 300 nM in a phosphate buffer (<xref ref-type="fig" rid="fig2">Figure 2C and D</xref>). This was also the case when the protein was diluted into 90 mM KCl (<xref ref-type="fig" rid="fig2">Figure 2E</xref>).</p><fig-group><fig id="fig2" position="float"><label>Figure 2.</label><caption><title>SRSF1 phase separation can be reduced using peptides that best mimic its RS repeats in their respective phosphorylation states.</title><p>(<bold>A</bold>) Schematic illustration of solubilizing phase-separating proteins using short peptides. Short peptides compete with RS repetitive regions, disrupting phase separation. (<bold>B</bold>) Domain architecture of SRSF1. The underlined serine residues in the SRSF1 RS domain can be phosphorylated, and the phosphorylated RS can be mimicked by ER and DR repeats. Short peptide co-solutes used in this study are shown below. (<bold>C</bold>) Phase separation of SRSF1. The left cuvette is SRSF1 solubilized in the RS8 peptide, and the right cuvette is SRSF1 in 140 mM potassium phosphate, pH 7.4, 10 mM NaCl. The fluorescence image of 288 nM unphosphorylated SRSF1 in phosphate buffer (<bold>D</bold>), KCl buffer (<bold>E</bold>). SRSF1 is labeled with Alexa488 at N220C. (<bold>F</bold>) SRSF1 solubility using 50 mM or 100 mM of peptide as indicated. (<bold>G</bold>) Ratio of solubility in peptides to solubility in 100 mM Arg/Glu as determined in panel D. (<bold>H</bold>) The RS8 peptide can reduce phase-separation droplets of SRSF1.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-84412-fig2-v2.tif"/></fig><fig id="fig2s1" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 1.</label><caption><title>Repetitive peptides solubilize other proteins containing similar repetitive regions.</title><p>(<bold>A</bold>) Solubilizing effects of 8-mer Arg-Ser (RS8), Glu-Arg (ER8), and Asp-Arg (DR8) were tested on Nob1 (blue bars) and Nop9 (green bars) at peptide concentrations of 50 mM (open bars) and 100 mM (filled bars). A buffer of 100 mM KCl was used as a control. (<bold>B</bold>) Nob1 protein sequence. (<bold>C</bold>) Nop9 protein sequence. Unstructured regions are shown in bold fonts. The basic and acidic-basic residue regions are in cyan and red, respectively.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-84412-fig2-figsupp1-v2.tif"/></fig></fig-group><p>To quantify protein solubility, we used ammonium sulfate precipitation followed by resuspension of the proteins in peptide-containing buffers (<xref ref-type="fig" rid="fig2">Figure 2F</xref>). This approach to measuring protein solubility has been used in many studies (<xref ref-type="bibr" rid="bib10">Burgess, 2009</xref>; <xref ref-type="bibr" rid="bib70">Trevino et al., 2008</xref>). If mimicking the repetitive sequences with the peptides helps resolve phase separation, we expect to see a more dramatic solubility increase when the peptide co-solutes most closely mimic the repetitive sequences of the proteins. To this end, we measured solubility of unphosphorylated full-length SRSF1, hyper-phosphorylated SRSF1 (pi-SRSF1), and RS-truncated SRSF1 (ΔRS) (<xref ref-type="fig" rid="fig2">Figure 2F</xref>). We found that whether phosphorylated or not, full-length SRSF1 was essentially insoluble in the 50 mM and 100 mM KCl control buffers (<xref ref-type="fig" rid="fig2">Figure 2F</xref>). Previous studies have found that truncation of the RS domain increases protein solubility (<xref ref-type="bibr" rid="bib69">Tacke and Manley, 1995</xref>). This suggests that the RS domain is responsible for the protein’s low solubility. To verify this, we measured the solubility of ΔRS and found that it was overall more soluble than full-length SRSF1 in all tested buffers (<xref ref-type="fig" rid="fig2">Figure 2F</xref>). Although an Arg/Glu mixture has been reported to promote protein solubility (<xref ref-type="bibr" rid="bib26">Golovanov et al., 2004</xref>), Arg/Glu at 100 mM provided only a limited solubilizing effect for full-length SRSF1 (<xref ref-type="fig" rid="fig2">Figure 2F</xref>). Among the peptides we tested, only RS8 dramatically increased solubility of unphosphorylated SRSF1 (from 0.6±0.29 μM in 100 mM KCl to 120±12 μM with 100 mM RS8). Consistent with our hypothesis, all tested peptides had a less dramatic solubilizing effect on ΔRS, likely because it does not contain the repeat sequences that the peptides are designed to mimic. Hyper-phosphorylation of an RS domain converts the region into a basic acidic repeat resembling ER and DR repeats. To mimic a phosphorylated RS domain, we tested the solubilizing effects of ER and DR peptides (<xref ref-type="bibr" rid="bib14">Cho et al., 2011</xref>; <xref ref-type="bibr" rid="bib24">Feng et al., 2012</xref>). ER8 increased the solubility of hyper-phosphorylated SRSF1 (pi-SRSF1) more than other peptides. DR8 and ER4 also had substantial, albeit less pronounced, solubilizing effects (<xref ref-type="fig" rid="fig2">Figure 2F</xref>). The preference for ER8 may be due to the fact that glutamic acid resembles phosphoserine more than aspartic acid in size. This confirmed our hypothesis that the solubilizing effect was more notable with peptides most closely resembling the proteins’ own repeats.</p><p>These trends are clearer when the effect of the peptides on solubility is normalized by the solubility in 100 mM Arg/Glu (<xref ref-type="fig" rid="fig2">Figure 2G</xref>). Whereas peptides designed to mimic the repetitive constructs produce as much as a 30-fold increase in solubility relative to Arg/Glu, the solubility increase of ΔRS only reaches about a twofold difference (<xref ref-type="fig" rid="fig2">Figure 2G</xref>). In accordance with our solubility tests, we found that using increasing concentrations of the RS8 peptide reduced the number of liquid-like droplets in solution (<xref ref-type="fig" rid="fig2">Figure 2H</xref>).</p><p>We also tested the solubilizing effect of these peptide co-solutes on two other RNA-binding proteins, Nob1 and Nop9. These peptides increased solubility of Nob1 and Nop9 by 50–60% and 2–10%, respectively (<xref ref-type="fig" rid="fig2s1">Figure 2—figure supplement 1A</xref>). Nob1 contains unstructured regions rich in basic and acidic-basic residues (<xref ref-type="fig" rid="fig2s1">Figure 2—figure supplement 1B</xref>) and has an increased solubility despite its unstructured regions having a lower homology to the tested peptides. Nop9 does not have such unstructured sequence regions (<xref ref-type="fig" rid="fig2s1">Figure 2—figure supplement 1C</xref>). Consistent with our hypothesis, the tested peptides have a moderate solubilizing effect on Nob1, whereas they have limited or no effect on Nop9. These results for SRSF1 constructs, Nob1, and Nop9 suggest that repetitive peptides improve solubility for proteins that have similar sequences.</p></sec><sec id="s2-3"><title>Peptide co-solutes are compatible with NMR experiments and binding assays</title><p>Ionic co-solutes usually increase the dielectric constant of a sample and complicate NMR data acquisition, producing difficulty in probe tuning/matching, elongation of pulse width, and reduction of sensitivity (<xref ref-type="bibr" rid="bib74">Wider and Dreier, 2006</xref>; <xref ref-type="bibr" rid="bib38">Kelly et al., 2002</xref>). This imposes a considerable obstacle, as NMR is one of the few methods that provides an atomic level description of the dynamic interactions of phase-separating proteins. This adverse effect can be quantified by the elongation of the pulse width, which is inversely proportional to the signal sensitivity (<xref ref-type="bibr" rid="bib74">Wider and Dreier, 2006</xref>). For example, increasing KCl concentration from 100 mM to 400 mM elongates the pulse width by about 50% on the NMR probe used in this study (<xref ref-type="fig" rid="fig3">Figure 3A</xref>). In contrast to the effect of KCl, peptide co-solutes did not significantly elongate the pulse width (<xref ref-type="fig" rid="fig3">Figure 3A</xref>). Whereas the 800 mM Arginine buffer used to solubilize SRSF1 during purification increased the pulse width to 16.97 µs, a combination of peptide and arginine had a less pronounced effect on the pulse width (<xref ref-type="fig" rid="fig3s1">Figure 3—figure supplement 1A</xref>). This is likely due to the low mobility of short peptides compared with salts or free amino acids (<xref ref-type="bibr" rid="bib38">Kelly et al., 2002</xref>).</p><fig-group><fig id="fig3" position="float"><label>Figure 3.</label><caption><title>Short peptides are compatible with NMR experiments.</title><p>(<bold>A</bold>) NMR 90 degree pulse width. ER8 is insoluble at 400 mM and therefore its pulse width could not be determined, as indicated by *. (<bold>B</bold>) <sup>15</sup>N-TROSY-HSQC overlay of SRSF1 in 100 mM RS8 and phosphorylated SRSF1 (pi-SRSF1) in 100 mM ER8. (<bold>C</bold>) Assigned residues in the SRSF1 protein sequence. Black bold fonts indicate non-overlapping residues. Gray fonts indicate unassigned residues. Color fonts indicate amino acids assigned to clusters. (<bold>D</bold>) Assignment of the SRSF1 amide groups.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-84412-fig3-v2.tif"/></fig><fig id="fig3s1" position="float" specific-use="child-fig"><label>Figure 3—figure supplement 1.</label><caption><title>The buffer 100 mM ER4, 400 mM Arg/Glu, pH 6.4 was effective at solubilizing and optimizing spectral quality for both unphosphorylated and hyperphosphorylated SRSF1 at high enough concentrations for NMR assignment.</title><p>(<bold>A</bold>) Pulse width of various concentrations of ER4 and Arg/Glu. (<bold>B</bold>) <sup>15</sup>N-TROSY-HSQC overlay of 370 µM SRSF1 (blue) and 540 µM phosphorylated SRSF1 (pi-SRSF1, red) in this buffer. (<bold>C</bold>) The buffer 100 mM ER4, 400 mM Arg/Glu, pH 6.4 did not abolish ligand binding. Fluorescence polarization assays for binding of SRSF1 constructs with a 5’ Alexa488-tagged RNA probe (UCAGAGGA). Both binding assays were performed in 100 mM ER4, 400 mM Arg/Glu, pH 6.5. Error bars represent SEM of three technical replicates.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-84412-fig3-figsupp1-v2.tif"/></fig></fig-group><p>We expect the peptides to compete with homotypic inter-molecular interactions (interactions between SRSF1 molecules) to solubilize the protein, but the competition should not be strong enough to abolish binding or disrupt protein structure. With the peptide co-solutes, we were able to obtain high quality NMR spectra for both unphosphorylated and phosphorylated SRSF1. The TROSY-HSQC overlay in <xref ref-type="fig" rid="fig3">Figure 3B</xref> is consistent with the expected presence of both globular domains that show a higher level of dispersion and disordered regions with proton shifts in the 8.0–8.5 ppm range. Although we were able to solubilize unphosphorylated and phosphorylated SRSF1 in these respective buffers (RS8 and ER8), a buffer that solubilizes both proteins was desired to allow direct comparison of the two proteins and facilitate NMR assignment. Therefore, we examined the spectra and pulse width of different combinations of peptides and Arginine (<xref ref-type="fig" rid="fig3s1">Figure 3—figure supplement 1</xref>). We found that a buffer of 100 mM ER4 mixed with 400 mM Arg/Glu, pH 6.4, maintained structure and solubilized both unphosphorylated and phosphorylated SRSF1 constructs at concentrations above 350 µM (<xref ref-type="fig" rid="fig3s1">Figure 3—figure supplement 1B</xref>). It also weakened but did not abolish binding of SRSF1 constructs to an RNA ligand (<xref ref-type="fig" rid="fig3s1">Figure 3—figure supplement 1C</xref>) and resulted in a pulse width of 15.04 µs (<xref ref-type="fig" rid="fig3s1">Figure 3—figure supplement 1A</xref>), significantly shorter than that observed for 800 mM Arg/Glu. This buffer was used for future experiments. Using this buffer, we assigned the backbone for unphosphorylated SRSF1 (<xref ref-type="fig" rid="fig3">Figure 3C and D</xref>). This accomplishment enabled us to investigate the mechanism by which repetitive peptides solubilize SRSF1. To this end, we selected RS8 and unphosphorylated SRSF1 for further study.</p></sec><sec id="s2-4"><title>Acidic and exposed aromatic residues of SRSF1 RRMs are responsible for the interactions with RS repeats that lead to phase separation</title><p>Mimic peptides were able to provide us with control over the critical point for SRSF1 phase separation, enabling us to obtain a backbone assignment in the solution state (<xref ref-type="fig" rid="fig3">Figure 3</xref>). According to our hypothesis, the mimic peptide should provide transient competition for contacts with the protein’s repetitive sequence. Inter- and intra- molecular interactions should still occur under these conditions. However, they should be weakened enough to prevent droplet formation, providing us with a stable sample that can be used to study the intermolecular interactions that lead to the initiation of phase separation. To verify that this was the case, we performed a series of paramagnetic relaxation enhancement NMR experiments to probe peptide, intramolecular, and homotypic intermolecular interactions.</p><p>To locate the RS-mimic peptide interacting sites, we labeled RS8 with a paramagnetic probe (MTSL) and mixed it with SRSF1 (<xref ref-type="fig" rid="fig4">Figure 4A and C</xref>, and <xref ref-type="fig" rid="fig4s1">Figure 4—figure supplement 1A</xref>). The paramagnetic probe decreases intensities of residue peaks on the NMR spectrum in a distance-dependent manner (<xref ref-type="bibr" rid="bib17">Clore and Iwahara, 2009</xref>). A higher PRE value indicates that peptides come closer to the residue analyzed. PRE is suitable for probing transient interactions, including the weak interactions between co-solutes and macromolecules (<xref ref-type="bibr" rid="bib17">Clore and Iwahara, 2009</xref>; <xref ref-type="bibr" rid="bib59">Okuno et al., 2021</xref>) and the intermolecular interactions that precede phase separation (<xref ref-type="bibr" rid="bib55">Murthy and Fawzi, 2020</xref>; <xref ref-type="bibr" rid="bib62">Ryan et al., 2018</xref>). We found that RS8 interacted primarily with RRM1 residues (<xref ref-type="fig" rid="fig4">Figure 4A and C</xref>, and <xref ref-type="fig" rid="fig4s1">Figure 4—figure supplement 1A</xref>). This is consistent with the fact that RRM1 (pI = 4.7) is more acidic than RRM2 (pI = 6.9), with regions of high negative charge on its two helices (<xref ref-type="fig" rid="fig4">Figure 4B</xref>). Dramatically perturbed sites were clustered on electronegative and aromatic sites, with the sequence D<sup>31</sup>IED on the α<sub>1</sub>-Helix and D<sup>44</sup>ID on the neighboring β-sheet being particularly perturbed (<xref ref-type="fig" rid="fig4">Figure 4C</xref>). Other hotspots included D<sup>80</sup>GYR and E<sup>87</sup>F on loops neighboring the α<sub>1</sub> and α<sub>2</sub> helices, respectively, D<sup>66</sup>AED on the α<sub>2</sub> helix, and RRM2 residues W134 and A<sup>150</sup>DVYR (<xref ref-type="fig" rid="fig4">Figure 4C</xref>).</p><fig-group><fig id="fig4" position="float"><label>Figure 4.</label><caption><title>SRSF1 residues involved in interactions with the RS8 peptide are similar to those found in intra-, and homotypic inter-molecular interactions with the RS region.</title><p>(<bold>A</bold>) <sup>15</sup>N-TROSY-HSQC overlay of SRSF1 in 50 mM diamagnetic (gray) and 2.5 mM paramagnetic RS8 (green). The intensities of residues close to the probe become diminished. Bleached residues (indicated by red type) came in such close contact with RS8 that their intensities were diminished before the first observation time point (additional information in the methods section). The full spectra are shown in <xref ref-type="fig" rid="fig4s1">Figure 4—figure supplement 1</xref>. (<bold>B</bold>) Electrostatic surface of SRSF1 RRM1 and RRM2. The α1 helix on RRM1 has a large negatively charged surface area, and RRM1 possesses overall more negative charge. (<bold>C</bold>) PRE values induced by 2.5 or 25 mM paramagnetic RS8. (<bold>D</bold>) Intra-molecular PRE produced by the MTSL-labeled RS region (N220C). (<bold>E</bold>) Inter-molecular PRE produced by the MTSL-labeled NMR-inactive SRSF1 (T248C). The filled symbols indicate bleached residues. Yellow sticks in the molecular graphics on the right indicate bleached residues. Gray indicates residues whose PRE values are unavailable due to peak overlap or an inability to assign them. PyMOL molecular graphics were prepared using Xplor-NIH (see Materials and methods section for more information).</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-84412-fig4-v2.tif"/></fig><fig id="fig4s1" position="float" specific-use="child-fig"><label>Figure 4—figure supplement 1.</label><caption><title>SRSF1 residues involved in interactions with the RS8 peptide are similar to those found in intra-, and homotypic inter-molecular interactions with the RS region.</title><p><sup>15</sup>N-TROSY-HSQC overlay of samples used in (<bold>A</bold>) Peptide-SRSF1, (<bold>B</bold>) Intramolecular, and (<bold>C</bold>) Intermolecular PRE experiments. Spectra in which the paramagnetic center was active are color coded, and spectra in which the paramagnetic center was quenched with ascorbic acid are in black. Residues marked in red came in close enough contact to paramagnetic centers that PRE could not be quantified. We refer to these residues as bleached. (<bold>D</bold>) Expanded view of a section of interest from the intermolecular PRE spectrum. (<bold>E</bold>) Expanded view of a section of interest in the intermolecular PRE spectrum.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-84412-fig4-figsupp1-v2.tif"/></fig><fig id="fig4s2" position="float" specific-use="child-fig"><label>Figure 4—figure supplement 2.</label><caption><title>SRSF1 residues involved in interactions with the RS8 peptide are similar to those found in intra-, and homotypic inter-molecular interactions with the RS region.</title><p>PRE of MTSL alone (<bold>A</bold>), an intramolecular PRE experiment with the MTSL tag on the very C-terminal end of the protein (<bold>B</bold>), and an intermolecular PRE experiment replicating the conditions of the intermolecular interactions experiment (<bold>C</bold>). When these spectra are subtracted from the experiments displayed in the main text (<bold>D–F</bold>) the trend is still similar.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-84412-fig4-figsupp2-v2.tif"/></fig></fig-group><p>According to our hypothesis, the RS8 peptide should provide transient competition with the RS domain without abolishing inter- and intra-molecular interactions (<xref ref-type="fig" rid="fig2">Figure 2A</xref>). To locate the intra-molecular interacting sites, we separately introduced the probe to the center of the RS domain (N220C, <xref ref-type="fig" rid="fig4">Figure 4D</xref>, <xref ref-type="fig" rid="fig4s1">Figure 4—figure supplement 1B, D</xref>) and the C-terminal end (T248C, <xref ref-type="fig" rid="fig4s1">Figure 4—figure supplement 1C, E</xref>). We found that labeling at the center of the RS domain produced the most notable perturbations (<xref ref-type="fig" rid="fig4">Figure 4D</xref>). To estimate the background PRE resulting from stochastic collisions, we also collected PRE data for SRSF1 mixed with the same concentration of probe alone as a control (<xref ref-type="fig" rid="fig4s2">Figure 4—figure supplement 2A</xref>). Subtracting the background PRE does not significantly change the perturbation pattern (<xref ref-type="fig" rid="fig4s2">Figure 4—figure supplement 2D and E</xref>). To verify that intermolecular interactions were not contributing to the measured intramolecular PRE, we performed a control PRE measurement by mixing an equal amount of probe-labeled (<sup>14</sup>N, N220C-MTSL) SRSF1 and <sup>15</sup>N-SRSF1 with no probe labeling (<xref ref-type="fig" rid="fig4s2">Figure 4—figure supplement 2C</xref>). Because <sup>14</sup>N SRSF1 cannot be detected by NMR HSQC, in this experiment, observed PRE can only happen through intermolecular interactions. At this total protein concentration of 200 µM, we did not see a significant intermolecular contribution to the PRE signal. The relative strengths of the spectra are readily observed when the intermolecular interactions under these conditions are subtracted from the intramolecular interactions (<xref ref-type="fig" rid="fig4s2">Figure 4—figure supplement 2F</xref>).</p><p>To locate the inter-molecular interacting sites, we placed the probe on the C-terminal end of the protein and doubled the concentration to a total of 400 µM protein, maintaining a 1:1 ratio of HSQC-undetectable SRSF1 with the probe attached and <sup>15</sup>N-labeled SRSF1 possessing no cysteines (<xref ref-type="fig" rid="fig4">Figure 4E</xref>, <xref ref-type="fig" rid="fig4s1">Figure 4—figure supplement 1C and E</xref>). With this experimental design, only inter-molecular interactions resulted in PRE on the <sup>15</sup>N-labeled SRSF1. Consistent with our hypothesis, the intra- and inter-molecular PRE patterns are similar to the perturbations from paramagnetic RS8.</p><p>To gain an atomic-level picture of the interactions between RS8 and SRSF1, we constructed models with the program Xplor-NIH using the PRE values as restraints (<xref ref-type="bibr" rid="bib64">Schwieters et al., 2006</xref>; <xref ref-type="bibr" rid="bib63">Schwieters et al., 2003</xref>) as illustrated in <xref ref-type="fig" rid="fig5s1">Figure 5—figure supplement 1</xref>. We further optimized these models using molecular dynamics simulations (<xref ref-type="fig" rid="fig5">Figure 5</xref>, <xref ref-type="fig" rid="fig5s2">Figure 5—figure supplement 2</xref>). The top 25% of initial Xplor-NIH structures agreed with the observed PRE data, with Pearson’s correlation coefficients of 0.916–0.941 (<xref ref-type="fig" rid="fig5s1">Figure 5—figure supplement 1</xref>). MD simulations produced structures in which a paramagnetic center was within the expected 12–15 Å from bleached residues (<xref ref-type="supplementary-material" rid="supp5">Supplementary file 5</xref>; <xref ref-type="bibr" rid="bib36">Iwahara et al., 2007</xref>). Representative images are displayed in <xref ref-type="fig" rid="fig5">Figure 5</xref>. The hotspot D<sup>31</sup>IED on the α<sub>1</sub> helix was found to be able to provide electrostatic contacts for multiple interactions, including hydrogen bonding with bleached isoleucine residues I42 and I45 (<xref ref-type="fig" rid="fig5">Figure 5A</xref>). In the region surrounding W134, an electrostatic interaction with D151 was found to enable a peptide arginine to orient parallel to the aromatic face of W134, forming cation-pi stacking interactions (<xref ref-type="fig" rid="fig5">Figure 5B</xref>). Simultaneous cation-pi stacking interactions were also observed in the regions surrounding D80 and Y79 (<xref ref-type="fig" rid="fig5">Figure 5C</xref>) as well as F88 (<xref ref-type="fig" rid="fig5">Figure 5D</xref>).</p><fig-group><fig id="fig5" position="float"><label>Figure 5.</label><caption><title>Electrostatic and cation-pi interactions are responsible for intermolecular interactions.</title><p>Molecular dynamics simulation of SRSF1 with four RS8 peptides.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-84412-fig5-v2.tif"/></fig><fig id="fig5s1" position="float" specific-use="child-fig"><label>Figure 5—figure supplement 1.</label><caption><title>Fitting of MD simulation structure to PRE values.</title><p>(<bold>A</bold>) Sample 10-membered ensemble generated by Xplor-NIH. RRM1 (residues 16–90) is held in place while the N-terminus and linker are allowed full flexibility (shown as transparent cartoons). RRM2 residues (residues 121–196) are allowed to move as a group (shown as transparent cartoons). Peptides are shown in green cartoons. (<bold>B</bold>) Sample correlation plot between observed PRE values and PRE values back calculated from the structural ensemble. All residues with for which PRE is obtained are included. Of the structures generated, the top 25% had Pearson’s correlation coefficients between 0.916 and 0.941 (R<sup>2</sup>=0.839–0.885). These were used to create an MD simulation starting structure in which a single peptide was bound to each hotspot.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-84412-fig5-figsupp1-v2.tif"/></fig><fig id="fig5s2" position="float" specific-use="child-fig"><label>Figure 5—figure supplement 2.</label><caption><title>MD simulation trajectories of SRSF1 ΔRS interacting with four RS8 peptides.</title><p>Root mean standard deviation between the simulation starting structure and the structure at a given simulation time (<bold>A–E</bold>) Interactions between peptides (left) and the SRSF1 ΔRS residues with which they interact (right). Without restraints, peptides interacting with loop regions (<bold>B, D</bold>) showed shorter-lived equilibria. In a separate simulation, a restraint was placed on peptide 2 (<bold>C</bold>) to maintain a single structure for observation. (<bold>F</bold>) RMSD of entire structure over the MD simulation period.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-84412-fig5-figsupp2-v2.tif"/></fig></fig-group></sec><sec id="s2-5"><title>SRSF1 RRM sites involved in phase separation are conserved in the SR protein family</title><p>We were curious whether interactions found in SRSF1 were conserved across the SR protein family. To this end, we used the program ClustalX (<xref ref-type="bibr" rid="bib46">Larkin et al., 2007</xref>) to align the RRM1 sequences of SR proteins and found that interaction hotspots on α1 and α2 helices of RRM1 domains were conserved in most SR proteins (<xref ref-type="fig" rid="fig6">Figure 6</xref>). The electronegative charge was most highly conserved in the α2 helix and the neighboring loop, while in the α1 helix, acidic residues were replaced by arginine residues in some SR proteins. RRM1 domains have a conserved RNA binding site (<xref ref-type="fig" rid="fig6">Figure 6</xref>, sticks). Interestingly, these regions are distal to the RNA-binding sites of the RRM domains, suggesting the conservation may be due to a role other than RNA recognition (<xref ref-type="fig" rid="fig6">Figure 6</xref>). Considering the conservation of these sites involved in phase separation, the phase-separating mechanism we revealed for SRSF1 could be applicable to many other members of the SR family.</p><fig id="fig6" position="float"><label>Figure 6.</label><caption><title>RRM1 residues responsible for SRSF1 phase separation are conserved throughout the SR protein family.</title><p>ClustalX alignment of RRM1 domains of the SR protein family, where yellow indicates identical amino acids, green and blue indicate conserved residues. Black boxes indicate PRE hotspots. Structure of RNA-bound SRSF1 RRM1 was obtained from PDB ID 6HPJ. Transparent electrostatic surface is displayed. Conserved electronegative residues opposite the RNA binding pocket are shown in sticks.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-84412-fig6-v2.tif"/></fig></sec></sec><sec id="s3" sec-type="discussion"><title>Discussion</title><p>There is currently a need to improve methods for determining which proteins phase separate and by what mechanism they do so, but solubility concerns make isolated experiments out of reach for many proteins. To solubilize phase-separating proteins, denaturants or high concentrations of salts are typically used. Denaturants are unsuitable for experiments characterizing native state proteins. High concentrations of salts are flawed as they interfere with NMR (<xref ref-type="bibr" rid="bib74">Wider and Dreier, 2006</xref>), SAXS (<xref ref-type="bibr" rid="bib60">Putnam et al., 2007</xref>), and circular dichroism (<xref ref-type="bibr" rid="bib27">Greenfield, 2006</xref>). Some proteins experience salting out when ionic co-solutes are introduced (<xref ref-type="bibr" rid="bib54">Murthy et al., 2019</xref>; <xref ref-type="bibr" rid="bib62">Ryan et al., 2018</xref>; <xref ref-type="bibr" rid="bib51">Martin et al., 2021</xref>). For example, we found that the solubility of SRSF1 is around 2–7 μM in 1–5 M of NaCl (data not shown).</p><p>The protein solubilizing strategy used here is of wide applicability, not just confined to the examples of SRSF1 and Nob1. Our bioinformatic search revealed that RS-containing proteins are highly abundant and that there is a positive correlation between these repeats and phase separation. In addition to RS repeats, GG, KK, QQ, PP, and RG demonstrated both a strong positive correlation between repeat length and phase separation and robust enough sample sizes to render these results significant (<xref ref-type="supplementary-material" rid="supp1">Supplementary file 1</xref>). However, it is important to note that the repeats mentioned above are likely not a comprehensive list of repetitive sequences conducive to phase separation. In fact, aside from LL, SG, PL, and TA repeats, all dipeptides that exist in 8-mer sequences possess at least a weak positive correlation between repeat length and tendency to appear in condensates (<xref ref-type="supplementary-material" rid="supp1">Supplementary file 1</xref>). It is also important to note that with our current knowledge, there is a possibility for both overestimation and underestimation of phase-separating proteins when using these databases. Some proteins that phase separate may not yet be identified. Further, proteins reported to be in condensates do not necessarily phase separate on their own. It is also possible that a protein can have more than one type of repeated motif mediating phase separation. For these reasons, in vitro phase separation experiments using purified protein are imperative. We hope that the use of this method will expand the number of techniques available to perform such experiments. Development of peptide structure-activity relationships (SARs) may serve as a technique for identifying the repeats responsible for phase separation as well. For instance, if a SAR reveals that a protein with more than one repetitive sequence reaches optimum solubility with one particular peptide mimic, the repetitive sequence corresponding to the peptide mimic may contribute more towards driving phase separation. Likewise, if a SAR reveals that a mixture of multiple mimic peptides is optimal for enhancing solubility, it is possible that multiple repetitive sequences within the protein are responsible for phase separation.</p><p>Characterization of intermolecular interactions that occur in the dispersed (soluble) state is an accepted method of understanding what interactions lead to phase separation (<xref ref-type="bibr" rid="bib55">Murthy and Fawzi, 2020</xref>; <xref ref-type="bibr" rid="bib62">Ryan et al., 2018</xref>; <xref ref-type="bibr" rid="bib20">Emmanouilidis et al., 2021</xref>). A previous comparison of intermolecular PRE spectra of the protein FUS in the dispersed versus the condensed (phase separated) state indicated that the transient intermolecular interactions seen between molecules in the solution state are comparable to the intermolecular contacts seen when the protein is phase separated (<xref ref-type="bibr" rid="bib54">Murthy et al., 2019</xref>; <xref ref-type="bibr" rid="bib53">Monahan et al., 2017</xref> as discussed in <xref ref-type="bibr" rid="bib55">Murthy and Fawzi, 2020</xref>). Our technique is unique in that it provides transient competition for the intermolecular interactions that lead to phase separation without abolishing these interactions entirely. We find that both RNA binding (<xref ref-type="fig" rid="fig3s1">Figure 3—figure supplement 1</xref>) and homotypic intermolecular interactions (<xref ref-type="fig" rid="fig4">Figure 4E</xref>) can still occur in a peptide-containing buffer. However, the presence of the peptide weakens intermolecular interactions enough to allow high quality NMR spectra to be obtained.</p><p>Here, we find that targeted competition for intermolecular interactions provides direct control over the critical point for phase separation, enabling experiments to be performed in the dispersed state that otherwise might not be possible. The ability to compare dispersed and condensed states for more proteins is still desirable. NMR spectra of isolated low complexity domains in the condensed state have been obtained successfully for several proteins including HNRNPA2, FUS, and a Caprin1-pFMRP complex (<xref ref-type="bibr" rid="bib54">Murthy et al., 2019</xref>; <xref ref-type="bibr" rid="bib62">Ryan et al., 2018</xref>; <xref ref-type="bibr" rid="bib75">Wong et al., 2020</xref>; <xref ref-type="bibr" rid="bib39">Kim et al., 2019</xref>; <xref ref-type="bibr" rid="bib11">Burke et al., 2015</xref>). However, interactions between the structured domains and low complexity domains of these proteins have not yet been probed using these techniques. One bottleneck to obtaining usable condensed state samples involves obtaining high concentrations of soluble protein before inducing phase separation in the sample (<xref ref-type="bibr" rid="bib54">Murthy et al., 2019</xref>; <xref ref-type="bibr" rid="bib62">Ryan et al., 2018</xref>; <xref ref-type="bibr" rid="bib75">Wong et al., 2020</xref>). Due to the high concentrations needed for an NMR backbone assignment and the negative effect of sample viscosity on NMR spectral quality, it is also more practical to perform backbone assignments of proteins in the dispersed states (<xref ref-type="bibr" rid="bib54">Murthy et al., 2019</xref>; <xref ref-type="bibr" rid="bib62">Ryan et al., 2018</xref>; <xref ref-type="bibr" rid="bib75">Wong et al., 2020</xref>). We hope our method may serve as a useful tool in expanding these techniques.</p><p>Homotypic intermolecular interactions and interactions with the peptides involve residues similar to those involved in intramolecular interactions. However, intermolecular interactions seem to have a greater preference for the more negatively charged RRM1. Whereas interactions with RRM1 appear at all concentrations studied, a concentration of 25 mM peptide is needed to observe bleaching of W134 and A150 on the electropositive RRM2 domain (<xref ref-type="fig" rid="fig4">Figure 4C</xref>). This difference may be due to the fact that, while intramolecular interactions involve an RS domain tethered to RRM2 that helps facilitate interaction, external RS repeats do not have a method of compensating for this charge repulsion.</p><p>This preference for RRM1 is interesting because the interactions seen on RRM2 involve the same residues that bind to RNA (<xref ref-type="bibr" rid="bib15">Cléry et al., 2013</xref>), but the interactions on RRM1 are opposite the RNA-binding interface (<xref ref-type="bibr" rid="bib16">Cléry et al., 2021</xref>). In fact, chemical shift perturbations performed in a previous study indicate that the α1 helical residues on RRM1, in particular, remain virtually unaffected when two different RNA ligands are introduced, indicating there are also no allosteric effects (<xref ref-type="bibr" rid="bib16">Cléry et al., 2021</xref>). Further, negative charges in this region are conserved across multiple SR protein family members (<xref ref-type="fig" rid="fig6">Figure 6</xref>). This suggests that RRM domains of SR proteins may have alternative sites used to mediate the protein-protein interactions that lead to phase separation. This is important because it means that the effect of phase separation on RNA binding can potentially be studied by disruption of these distal sites.</p><p>As we learn more about biomolecular condensates, it is of interest to understand what causes proteins to migrate to one condensate over another. It has been shown that the isolated SRSF2 RRM can localize to speckles on its own (<xref ref-type="bibr" rid="bib28">Greig et al., 2020</xref>), which suggests there may be an additional molecular grammar within the structured components of these proteins that directs them towards nuclear speckles. It is known that speckles rely on RS repeats as scaffolds, as truncation of SRRM2’s regions containing RS repeats (<xref ref-type="fig" rid="fig1s1">Figure 1—figure supplement 1</xref>) disrupts speckles (<xref ref-type="bibr" rid="bib34">Ilik et al., 2020</xref>; <xref ref-type="bibr" rid="bib77">Xu et al., 2022</xref>). It is possible that these RS repeats function in part by providing multiple interaction sites for this type of RRM.</p><sec id="s3-1"><title>Ideas and speculation</title><p>We demonstrate that electronegative α<sub>1</sub> and α<sub>2</sub> helices along with neighboring aromatic residues serve as interacting sites for unphosphorylated RS repetitive sequences. This finding has implications for how phosphorylation might change interactions within the speckles. If negatively charged residues are important for maintaining protein-protein interactions, charge repulsion between a hyperphosphorylated tail and acidic residues might be one reason that SR proteins leave the speckles upon hyperphosphorylation (<xref ref-type="bibr" rid="bib29">Gui et al., 1994</xref>). Phosphoserines of RS repeats have been proposed to form salt bridges with neighboring arginine residues (<xref ref-type="bibr" rid="bib30">Hamelberg et al., 2007</xref>), although any such contacts are likely short-lived and do not result in a stable secondary structure (<xref ref-type="bibr" rid="bib57">Ngo et al., 2008</xref>; <xref ref-type="bibr" rid="bib76">Xiang et al., 2013</xref>). Temporary phosphoserine-arginine contacts may be sufficient to compete with the highly transient cation-pi stacking interactions that we observe here.</p></sec></sec><sec id="s4" sec-type="materials|methods"><title>Materials and methods</title><sec id="s4-1"><title>SRSF1 expression and purification</title><p>The DNA encoding human SRSF1 was sub-cloned into pSMT3 using BamH I and Hind III. The ΔRS construct and mutants SRSF1 C16S C148S N220C (N220C), SRSF1 C16S C148S T248C (T248C), and SRSF1 C16S C148S (NoC) were prepared using mutagenesis PCR. All these mutants maintain the folded structure according to NMR spectra, and bind with SRSF1 cognate RNA UCAGAGGA. All proteins were expressed by BL21-CodonPlus (DE3) cells in LB media or minimal media supplemented with proper isotopes for NMR experiments. Hyperphosphorylated SRSF1 was prepared by co-transformation of BL21-CodonPlus (DE3) cells using pSMT3/SRSF1 and CDC2-like kinase 1 (CLK1) cloned in pETDuet-1. Cells were cultured at 37 °C to reach an OD600 of 0.6, and 0.5 mM IPTG was added to induce protein expression. Cells were further cultured 16 hours at 22 °C. The cells were harvested by centrifugation (4000 RCF, 15 min). The cell pellet was re-suspended in 20 mM HEPES, pH 7.5, 150 mM Arg/Glu, 2 M NaCl, 25 mM imidazole, 0.2 mM TCEP supplemented with 1 mM PMSF, 1 mg/mL lysozyme, 1 tablet of Pierce protease inhibitor, and 1 mM NaVO4 for the hyperphosphorylated construct. After three freeze-thaw cycles, the sample was sonicated and centrifuged at 23,710 g for 40 min using a Beckman Coulter Avanti JXN26/JA20 centrifuge. The supernatant was loaded onto 5 mL of HisPur Nickel-NTA resin and then eluted with 60 mL of 20 mM MES pH 6.5, 300 mM imidazole, 600 mM Arg/Glu, and 0.2 mM TCEP. The eluted sample was cleaved with 2 µg/mL Ulp1 for 2 hr at 37 °C. The four unphosphorylated SRSF1 constructs (WT, N220C, T248C, NoC) were further purified by a 5 mL HiTrap Heparin column. The hyperphosphorylated SRSF1 was further purified by a 5 mL Cytiva Fast Flow Q column. The eluted samples from the ion exchange step were further purified by a HiLoad 16/60 Superdex 75 pg size exclusion column equilibrated with 800 mM Arg/Glu, pH 6.5, 1 mM TCEP, 0.02% NaN<sub>3</sub>. The protein identities were confirmed by mass spectrometry. As reported in previous study, 18-phosphates were added on the RS region of SRSF1 (<xref ref-type="bibr" rid="bib2">Aubol et al., 2013</xref>). The protein purities were judged to be &gt;95% based on SDS-PAGE.</p><p>Between purification and NMR experiments, the protein was transferred to peptide buffer in one of three ways: (1) It was concentrated in 800 mM Arg/Glu pH 6.5, 1 mM TCEP, 0.02% NaN<sub>3</sub> and diluted with a peptide buffer to the final concentration (For <xref ref-type="fig" rid="fig3">Figures 3D</xref> and <xref ref-type="fig" rid="fig4">4D–E</xref>, <xref ref-type="fig" rid="fig3s1">Figure 3—figure supplement 1</xref>, <xref ref-type="fig" rid="fig4s1">Figure 4—figure supplement 1</xref>, <xref ref-type="fig" rid="fig4s2">Figure 4—figure supplement 2</xref>). (2) It was precipitated and re-suspended in the peptide buffer (for <xref ref-type="fig" rid="fig3">Figure 3B</xref>). (3) For the peptide titrations (<xref ref-type="fig" rid="fig4">Figure 4C</xref>), because a low concentration was needed, the initial spectrum was taken in 200 mM Arg/Glu, and peptide was titrated into the NMR tube.</p></sec><sec id="s4-2"><title>Nop9 and Nob1 expression and purification</title><p>Nop9 and Nob1 expression and purification are detailed in published papers (<xref ref-type="bibr" rid="bib80">Zhang et al., 2016</xref>; <xref ref-type="bibr" rid="bib44">Lamanna and Karbstein, 2009</xref>). SUMO-tagged proteins were induced by 0.4 mM IPTG and expressed at 22 °C overnight in <italic>E. coli</italic> strain BL21-CodonPlus (DE3). The LB miller medium was supplemented with 0.1 mM ZnSO<sub>4</sub> for Nob1 expression. Cell pellets were re-suspended in 25 mM HEPES, pH 7.5, 1 M NaCl, 1 mM TCEP, 25 mM imidazole, 1 mg/mL lysozyme and lysed by sonication, followed by centrifugation. The supernatant was applied to HisPur Ni-NTA resin, washed with 200 mL of loading buffer, and eluted with 25 mM HEPES, pH 7.5, 500 mM NaCl, 1 mM TCEP, 500 mM imidazole. The SUMO tag was cleaved overnight with 2 µg/mL of Ulp1 at 4 °C. The cleaved sample was purified by a 5 mL HiTrap Heparin column (GE Healthcare), and polished using a HiLoad 16/60 Superdex 200 column (GE Healthcare) equilibrated in 25 mM HEPES, pH 7.5, 500 mM NaCl, and 1 mM TCEP. The protein purities were &gt;95% based on SDS-PAGE.</p></sec><sec id="s4-3"><title>NMR assignment</title><p>SRSF1 cultured in <sup>2</sup>H,<sup>13</sup>C, <sup>15</sup>N M9 media was concentrated to 370 µM in 100 mM ER4, 400 mM Arg/Glu, pH 6.4, 1 mM TCEP, 10% D<sub>2</sub>O, and 0.02% NaN<sub>3</sub>. Triple resonance assignment experiments HNCA, HNCACB, HN(CO)CA, HN(CO)CACB, HNCO, and HN(CA)CO were collected at 37 °C on a Bruker Avance III-HD 850 MHz spectrometer installed with a cryo-probe. Approximately 85% of the protein backbone region was assigned using this method. Another approximately 13% of backbone exists in the disordered state with highly degenerate sequences, which leads to heavy peak overlap. These RS and G-rich regions were grouped into clusters. Multiplicity selective in-phase coherence transfer (MUSIC) experiments were collected to further characterize the clusters and verify the assignment of the rest of the protein. MUSIC was performed on SRSF1 for the following amino acids: Ser, Arg, Thr, Asn, Ala, Tyr/His/Phe, Pro, Asn/Gln, Met, and Gly. When used in combination with analysis of the effect of paramagnetic tag placement, peak clusters were able to be assigned to locations on the disordered regions. The NMR data was processed using NMRPipe (<xref ref-type="bibr" rid="bib18">Delaglio et al., 1995</xref>), and assignment was performed using NMRViewJ (<xref ref-type="bibr" rid="bib37">Johnson, 2004</xref>). The assignment of the well dispersed regions (85% of the protein) has been submitted to BMRB (ID: 51299).</p></sec><sec id="s4-4"><title>Paramagnetic relaxation enhancement (PRE) measurements</title><p>RS peptide with a sequence ‘SRSRSRSRC’ was synthesized and purified by GenScript with a purity &gt;98%. The cysteine residue at the C-terminus was introduced for MTSL labeling. RS peptide was mixed with MTSL in a molar concentration ratio of 1:4. The pH was adjusted to 7.0 before a 2 hr labeling at room temperature. To remove unreacted MTSL, 10 mL of ether was added to the sample, and the mixture was vortexed and spun at 4000 rpm for 5 min. The extraction process was repeated twice. After purification, the pH of the peptide was adjusted to 6.5 using KOH, and the peptide was lyophilized. <sup>1</sup>H paramagnetic relaxation enhancement (PRE) data was gathered at 37 °C on a Bruker Avance III-HD 850 MHz spectrometer installed with a cryo-probe. Titrations were performed by adding solid peptide to a <sup>15</sup>N-labeled SRSF1 construct without cysteine (the NoC construct) in 200 mM Arg/Glu, pH 6.3, 1 mM TCEP. PRE spectra were obtained for the protein in 200 mM Arg/Glu alone, 200 mM Arg/Glu with 2.5 mM peptide, 200 mM Arg/Glu with 25 mM peptide, and 200 mM Arg/Glu with 50 mM peptide. After the spectrum with 50 mM peptide was collected, MTSL was quenched using 10 mM sodium ascorbate, and a PRE experiment was run on the quenched sample.</p><p>To prepare the sample for inter-molecular PRE, an SRSF1 construct with one cysteine at the C-terminal end of the RS tail (T248C) was obtained using mutagenesis PCR, and the mutated protein was expressed by BL21-CodonPlus (DE3) cells in LB media. The protein was exchanged into an MTSL labeling buffer of 0.8 M Arg/Glu, 100 mM NaCl, 50 mM Tris-HCl pH 7.0 using a HiPrep 26/10 desalting column. The sample was diluted to a concentration of 20 µM, and MTSL was added to a concentration of 0.4 mM. The sample was incubated in the dark at 37 °C for 12 hr, after which a desalting column was used to remove unreacted MTSL. The MTSL-labeled, NMR-inactive SRSF1 was mixed in a 1:1 ratio with a <sup>15</sup>N SRSF1 construct with no cysteine (NoC) in a buffer of 100 mM ER4, 5 mM MES, pH 6.4, 400 mM Arg/Glu, and 5% D2O. The final concentration of protein was 420 µM (210 µM SRSF1 C16/148 S T248C-MTSL and 210 µM <sup>15</sup>N SRSF1 C16/148 S).</p><p>To measure intramolecular PRE, an SRSF1 construct with one cysteine at the center of the first RS domain (SRSF1 C16S/C148S/N220C) was obtained using mutagenesis PCR and the purification method described above with the growth performed in M9 media containing <sup>15</sup>N isotopes. MTSL was labeling was performed as described above. The final concentration of protein was 220 µM.</p><p>The low concentration intermolecular PRE was collected between SRSF1 C16/148 S N220C-MTSL and <sup>15</sup>N SRSF1 C16/148 S at a total concentration of 185 µM (93 µM of each construct) as described above. A control PRE experiment with free MTSL was conducted by adding 220 µM MTSL to 220 µM <sup>15</sup>N SRSF1 C16/C148S (NoC) in the NMR buffer described above.</p><p>All PRE measurements were carried out using a pulse sequence developed by Junji Iwahara (<xref ref-type="bibr" rid="bib36">Iwahara et al., 2007</xref>). Diamagnetic data were collected after adding 2 mM ascorbic acid. The NMR data was processed using NMRPipe (<xref ref-type="bibr" rid="bib18">Delaglio et al., 1995</xref>) and analyzed using NMRViewJ (<xref ref-type="bibr" rid="bib37">Johnson, 2004</xref>). PRE values and errors were estimated as described previously (<xref ref-type="bibr" rid="bib36">Iwahara et al., 2007</xref>).</p><p>Residues were considered above noise level if their intensities at the second time point on the diamagnetic spectrum were greater than five times the standard deviation of the spectrum. Peaks below this noise level threshold were excluded from analysis on all spectra. Peaks in regions of high spectral overlap were also excluded. If at the first time point of collection (approximately 12ms after the first 90<sup>o</sup> pulse), the intensity of the paramagnetic peak was less than or equal to half of the intensity of the diamagnetic peak at that time point, the peaks were was defined as bleached. In these circumstances, the relaxation occurred too quickly to allow fitting of the exponential decay curve. Residues G52 and R154 met the definition of both bleached and noisy. They were excluded from analysis.</p></sec><sec id="s4-5"><title>Solubility assays</title><p>Purified protein aliquots (40 μL) were incubated with 3.2 M ammonium sulfate on ice for 30 min before 10-min centrifugation at 14,000 RCF at 4 °C. After confirming that no protein was present in the supernatant, the supernatant was discarded. The pellets were re-suspended in 20 µL of corresponding buffers and shaken at room temperature for 30 min. The re-suspensions were further centrifuged at room temperature at 14,000 RCF for 5 min. The protein concentrations in supernatants were measured using UV absorbance at 280 nm. Error bars represent standard deviation from three technical repeats. The initial concentration of full-length SRSF1 constructs was 250 μM. As RS-deleted SRSF1 has a higher solubility, the initial protein concentration used for this construct was 400 μM.</p></sec><sec id="s4-6"><title>Molecular graphics</title><p>An Alphafold structure was downloaded from the Uniprot website for SRSF1 ΔRS and refined using Xplor-NIH. Restraints used for refinement included dihedral angles obtained from the assignment, RDC values obtained in a previous study (<xref ref-type="bibr" rid="bib21">Fargason et al., 2020</xref>), and chemical shift perturbations. PRE values were projected onto to the structure in PyMOL by reassigning B-factors and coloring on a ramp scale.</p></sec><sec id="s4-7"><title>MD simulations</title><p>An Alphafold structure was downloaded from the Uniprot website for SRSF1 ΔRS and refined using Xplor-NIH. Docking of peptides was accomplished with Xplor-NIH using a restrained rigid-body simulated annealing protocol refined against the PRE, CSP, RDC, and dihedral angle data. In total, 100 Xplor-NIH structures were calculated using an ensemble size of 10. Each ensemble member had a single peptide, resulting in 10 total peptides in the model. The RRM1 domain (residues 16–90) was held in place while the N-terminus and linker were allowed full flexibility. RRM2 residues (residues 121–196) were allowed to move as a group. Linker and N-terminal residues were allowed full flexibility.</p><p>The general distance relationship for PRE is defined as <xref ref-type="bibr" rid="bib36">Iwahara et al., 2007</xref>:<disp-formula id="equ1"><label>(1)</label><mml:math id="m1"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:msub><mml:mi mathvariant="normal">Γ</mml:mi><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:msup><mml:mi>r</mml:mi><mml:mrow><mml:mn>6</mml:mn></mml:mrow></mml:msup></mml:mfrac><mml:msup><mml:mrow><mml:mo>(</mml:mo><mml:mfrac><mml:msub><mml:mi>μ</mml:mi><mml:mrow><mml:mn>0</mml:mn></mml:mrow></mml:msub><mml:mrow><mml:mn>4</mml:mn><mml:mi>π</mml:mi></mml:mrow></mml:mfrac><mml:mo>)</mml:mo></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup><mml:mfrac><mml:mn>1</mml:mn><mml:mn>15</mml:mn></mml:mfrac><mml:msubsup><mml:mi>γ</mml:mi><mml:mrow><mml:mi>I</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msubsup><mml:msup><mml:mi>g</mml:mi><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup><mml:msubsup><mml:mi>μ</mml:mi><mml:mrow><mml:mi>B</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msubsup><mml:mi>S</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>S</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn><mml:mo stretchy="false">)</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mn>4</mml:mn><mml:msub><mml:mi>τ</mml:mi><mml:mrow><mml:mi>c</mml:mi></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:mfrac><mml:mrow><mml:mn>3</mml:mn><mml:msub><mml:mi>τ</mml:mi><mml:mrow><mml:mi>c</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:mn>1</mml:mn><mml:mo>+</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>ω</mml:mi><mml:mrow><mml:mi>H</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mi>τ</mml:mi><mml:mrow><mml:mi>c</mml:mi></mml:mrow></mml:msub><mml:msup><mml:mo stretchy="false">)</mml:mo><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup></mml:mrow></mml:mfrac></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mstyle></mml:mrow></mml:math></disp-formula></p><p>where Γ<sub>2</sub> is the PRE value, r is the distance between the paramagnetic center and the observed nucleus, µ<sub>0</sub> is the vacuum permeability constant, γ<sub>I</sub> is the nuclear gyromacnetic ratio, g is the electron g-factor, µ<sub>B</sub> is the electron Bohr magneton, S is the electron spin quantum number, ω<sub>H</sub>/2π is the nuclear Larmor frequency, and τ<sub>c</sub> is the PRE correlation time (where τ<sub>c</sub><sup>–1</sup> = τ<sub>r</sub><sup>–1</sup>+ τ<sub>s</sub><sup>–1</sup>, τ<sub>r</sub> = nuclear rotational correlation time, τ<sub>s</sub> = electron relaxation time).</p><p>Because of the flexible nature of the protein, the structures were described using multiple ensemble members. For each residue (<italic>h</italic>), the PRE was determined by the average distance across the ensemble members:<disp-formula id="equ2"><label>(2)</label><mml:math id="m2"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:msub><mml:mrow><mml:mo>⟨</mml:mo><mml:msup><mml:mi>r</mml:mi><mml:mrow><mml:mo>−</mml:mo><mml:mn>6</mml:mn></mml:mrow></mml:msup><mml:mo>⟩</mml:mo></mml:mrow><mml:mrow><mml:mi>h</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mi>N</mml:mi></mml:mfrac><mml:munderover><mml:mo movablelimits="false">∑</mml:mo><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>N</mml:mi></mml:mrow></mml:munderover><mml:mrow><mml:msubsup><mml:mi>r</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mo>−</mml:mo><mml:mn>6</mml:mn></mml:mrow></mml:msubsup></mml:mrow></mml:mstyle></mml:mrow></mml:math></disp-formula></p><p>where N is the number of ensemble members, and r<sub>i</sub> is the distance between the paramagnetic center (the nitroxy oxygen of MTSL) and the nucleus under observation (the amide proton) in a single ensemble. Angle brackets (&lt;&gt;) indicate ensemble averages. The PRE values were back-calculated using the SBMF mode described in <xref ref-type="bibr" rid="bib35">Iwahara et al., 2004</xref>:</p><p>Agreement between experimentally observed PRE and back-calculated PRE was assessed using the Q-factor (Q) and Pearson Correlation coefficients (R):<disp-formula id="equ3"><label>(3)</label><mml:math id="m3"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mi>Q</mml:mi><mml:mo>=</mml:mo><mml:msqrt><mml:mfrac><mml:mrow><mml:munderover><mml:mo movablelimits="false">∑</mml:mo><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:munderover><mml:mrow><mml:msup><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:msubsup><mml:mi mathvariant="normal">Γ</mml:mi><mml:mrow><mml:mn>2</mml:mn></mml:mrow><mml:mrow><mml:mi>o</mml:mi><mml:mi>b</mml:mi><mml:mi>s</mml:mi></mml:mrow></mml:msubsup><mml:mo stretchy="false">(</mml:mo><mml:mi>h</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mo>−</mml:mo><mml:msubsup><mml:mi mathvariant="normal">Γ</mml:mi><mml:mrow><mml:mn>2</mml:mn></mml:mrow><mml:mrow><mml:mi>c</mml:mi><mml:mi>a</mml:mi><mml:mi>l</mml:mi><mml:mi>c</mml:mi></mml:mrow></mml:msubsup><mml:mo stretchy="false">(</mml:mo><mml:mi>h</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mo>}</mml:mo></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup></mml:mrow></mml:mrow><mml:mrow><mml:munderover><mml:mo movablelimits="false">∑</mml:mo><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:munderover><mml:mrow><mml:msubsup><mml:mi mathvariant="normal">Γ</mml:mi><mml:mrow><mml:mn>2</mml:mn></mml:mrow><mml:mrow><mml:mi>o</mml:mi><mml:mi>b</mml:mi><mml:mi>s</mml:mi></mml:mrow></mml:msubsup><mml:mo stretchy="false">(</mml:mo><mml:mi>h</mml:mi><mml:msup><mml:mo stretchy="false">)</mml:mo><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup></mml:mrow></mml:mrow></mml:mfrac></mml:msqrt></mml:mstyle></mml:mrow></mml:math></disp-formula><disp-formula id="equ4"><label>(4)</label><mml:math id="m4"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mi>R</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:munderover><mml:mo movablelimits="false">∑</mml:mo><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:munderover><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:msubsup><mml:mi mathvariant="normal">Γ</mml:mi><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mi>o</mml:mi><mml:mi>b</mml:mi><mml:mi>s</mml:mi></mml:mrow></mml:msubsup><mml:mo>−</mml:mo><mml:mover><mml:msup><mml:mi mathvariant="normal">Γ</mml:mi><mml:mrow><mml:mi>o</mml:mi><mml:mi>b</mml:mi><mml:mi>s</mml:mi></mml:mrow></mml:msup><mml:mo accent="false">¯</mml:mo></mml:mover><mml:mo stretchy="false">)</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:msubsup><mml:mi mathvariant="normal">Γ</mml:mi><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mi>c</mml:mi><mml:mi>a</mml:mi><mml:mi>l</mml:mi><mml:mi>c</mml:mi></mml:mrow></mml:msubsup><mml:mo>−</mml:mo><mml:mover><mml:msup><mml:mi mathvariant="normal">Γ</mml:mi><mml:mrow><mml:mi>c</mml:mi><mml:mi>a</mml:mi><mml:mi>l</mml:mi><mml:mi>c</mml:mi></mml:mrow></mml:msup><mml:mo accent="false">¯</mml:mo></mml:mover><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mrow><mml:msqrt><mml:munderover><mml:mo movablelimits="false">∑</mml:mo><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:munderover><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:msubsup><mml:mi mathvariant="normal">Γ</mml:mi><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mi>o</mml:mi><mml:mi>b</mml:mi><mml:mi>s</mml:mi></mml:mrow></mml:msubsup><mml:mo>−</mml:mo><mml:mover><mml:msup><mml:mi mathvariant="normal">Γ</mml:mi><mml:mrow><mml:mi>o</mml:mi><mml:mi>b</mml:mi><mml:mi>s</mml:mi></mml:mrow></mml:msup><mml:mo accent="false">¯</mml:mo></mml:mover><mml:msup><mml:mo stretchy="false">)</mml:mo><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup></mml:mrow><mml:msqrt><mml:munderover><mml:mo movablelimits="false">∑</mml:mo><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:munderover><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:msubsup><mml:mi mathvariant="normal">Γ</mml:mi><mml:mrow><mml:mi>h</mml:mi></mml:mrow><mml:mrow><mml:mi>c</mml:mi><mml:mi>a</mml:mi><mml:mi>l</mml:mi><mml:mi>c</mml:mi></mml:mrow></mml:msubsup><mml:mo>−</mml:mo><mml:mover><mml:msup><mml:mi mathvariant="normal">Γ</mml:mi><mml:mrow><mml:mi>c</mml:mi><mml:mi>a</mml:mi><mml:mi>l</mml:mi><mml:mi>c</mml:mi></mml:mrow></mml:msup><mml:mo accent="false">¯</mml:mo></mml:mover><mml:msup><mml:mo stretchy="false">)</mml:mo><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup></mml:mrow></mml:msqrt></mml:msqrt></mml:mfrac></mml:mstyle></mml:mrow></mml:math></disp-formula></p><p>where n is the number of residues for which PRE values were obtained.</p><p>The top 25% of structures possessed Pearson correlation coefficients between 0.916 and 0.941 and Q-factors between 0.454 and 0.548.</p><p>These Xplor-NIH structures were used to produce an MD-simulation starting structure with 4 peptides and 1 SRSF1 ΔRS structure. The structure was further refined using AMBER20 with the ff19SB forcefield. Solvation was performed with explicit TIP3P water molecules with 0.15 M NaCl used to balance the charges. The simulation temperature was set to 300 K, and the cutoff distance of nonbonded interactions was set to 10 Å. A simulation in which no restraints were applied was run for 201 ns. This simulation accounted for bleached residues, which, with the exception of residue 88, remained within the expected 12–15 Å from the paramagnetic probe. For residue 88, a separate simulation was run for 17 ns in which a distance restraint was used. The distance restraint maintained an interaction between F88 and peptide 2 that was generated by Xplor-NIH and consistent with the bleaching on the PRE spectrum. Distance restraints were not applied to other sites. Whereas the 10 peptides in the Xplor-NIH models accounted for all PRE data, the four peptides were only sufficient to cover the hotspot regions. For the remaining residues on the α<sub>1</sub> and α<sub>2</sub> helices, four peptides only partially accounted for PRE data (<italic>R</italic>=0.681 for residues 29–38 and 64–74), and four peptides and was not sufficient to account for the PRE values across the molecule as a whole (<italic>R</italic>=0.136).</p><p>During the simulation, the nature of interactions within the hotspots changed in some cases, but the distance between peptides and bleached residues did not significantly change. The MD trajectory analysis was performed by CPPTRAJ.</p></sec><sec id="s4-8"><title>Bioinformatics analysis</title><p>Domain annotations and sequences for human proteins were obtained from the Uniprot website. Analysis was restricted to full-length, reviewed, human proteins for which there was evidence at the protein level. A Python script was used to search for consecutive Ser-Arg or Arg-Ser repeats of 4, 6, or 8 amino acids. Identification of proteins in condensates was based on databases PhaSepDB (PhaSepDB2.0 download), DrLLPs, and LLPSDB (natural protein download).</p><p>RRM domains were identified by further restricting our Uniprot search to proteins containing RRM domains of any manual assertion. In analysis of the correlation between condensation, RS repeats, and RRM domains, we found 206 proteins that together contained 365 RRM domains. Python scripts and bioinformatic data can be accessed via the github link: <ext-link ext-link-type="uri" xlink:href="https://github.com/taliafargason/Repeats_in_Condensates">https://github.com/taliafargason/Repeats_in_Condensates</ext-link> (<xref ref-type="bibr" rid="bib22">Fargason, 2023</xref>).</p><p>Percent composition values were obtained using the program LCD composer developed by Sean Cascarina in the Ross lab (<xref ref-type="bibr" rid="bib12">Cascarina et al., 2021</xref>). Search was conducted for sequences 20 amino acids in length with at least 5% composition R/S. Proteins were then crossmatched against the lists of proteins containing RRM domains and proteins found in condensates. Proteins with more than one hit were only counted once (the sequence with the highest percent composition was used).</p></sec><sec id="s4-9"><title>Statistical analysis of the effect of peptide length on phase separation (<xref ref-type="supplementary-material" rid="supp1">Supplementary file 1</xref>)</title><p>Five categories of proteins were identified: proteins with no instances of the dipeptide (x=0), proteins with at least one instance of the dipeptide (x&gt;2), proteins with at least one 4-mer dipeptide repeat (x&gt;4), proteins with at least one 6-mer dipeptide repeat (x&gt;6), and proteins with at least one 8-mer dipeptide repeat (x&gt;8). Only proteins in category x=0 were automatically excluded from other categories. For instance, if a protein had a 16-mer RG repeat, it would be counted in the x&gt;2, x&gt;4, x&gt;6, and x&gt;8 categories but not x=0. Likewise, if a protein had neither ‘RG’ nor ‘GR’ anywhere in its sequence, it would be counted in the x=0 category only. Only the length of the longest uninterrupted repeat was considered. The number of repeats in the protein was not a factor in this analysis.</p><p>Within each category, the fraction of protein in condensates (<italic>f<sub>ps</sub></italic>)was calculated as:<disp-formula id="equ5"><label>(5)</label><mml:math id="m5"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:msub><mml:mi>f</mml:mi><mml:mrow><mml:mi>p</mml:mi><mml:mi>s</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mfrac><mml:msub><mml:mi>N</mml:mi><mml:mrow><mml:mi>p</mml:mi><mml:mi>s</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:msub><mml:mi>N</mml:mi><mml:mrow><mml:mi>p</mml:mi><mml:mi>s</mml:mi></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mi>N</mml:mi><mml:mrow><mml:mi>n</mml:mi><mml:mi>p</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfrac></mml:mstyle></mml:mrow></mml:math></disp-formula></p><p>where <italic>N<sub>ps</sub></italic> is the number of proteins with the repeat that have been found in condensates and <italic>N<sub>np</sub></italic> is the number of proteins with the repeat that have not been found in condensates.</p><p>A population-based error (<italic>E<sub>ps</sub></italic>) was calculated as:<disp-formula id="equ6"><label>(6)</label><mml:math id="m6"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:msub><mml:mi>E</mml:mi><mml:mrow><mml:mi>p</mml:mi><mml:mi>s</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:msqrt><mml:msub><mml:mi>N</mml:mi><mml:mrow><mml:mi>p</mml:mi><mml:mi>s</mml:mi></mml:mrow></mml:msub></mml:msqrt></mml:mfrac></mml:mstyle></mml:mrow></mml:math></disp-formula></p><p>Dipeptide repeats were considered within this error if they existed in the 8-mer form and met the criterion:<disp-formula id="equ7"><label>(7)</label><mml:math id="m7"><mml:mrow><mml:msub><mml:mi>E</mml:mi><mml:mrow><mml:mi>p</mml:mi><mml:mi>s</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo>≥</mml:mo><mml:mn>8</mml:mn><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:msub><mml:mo>&lt;</mml:mo><mml:mn>2</mml:mn><mml:mo>∗</mml:mo><mml:msub><mml:mi>f</mml:mi><mml:mrow><mml:mi>p</mml:mi><mml:mi>s</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo>≥</mml:mo><mml:mn>8</mml:mn><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:msub></mml:mrow></mml:math></disp-formula></p><p>A correlation analysis was performed between the number of repeats (x) and the fraction of proteins in condensates (<italic>f<sub>ps</sub></italic>). Dipeptide repeats were considered to correlate significantly with phase separation if they met the criteria of: <italic>R</italic>&gt;0 (positive correlation) and p&lt;0.05.</p></sec><sec id="s4-10"><title>Statistical analysis on the effect of RS repeats and RRM domains on phase separation (<xref ref-type="fig" rid="fig1">Figure 1</xref> and <xref ref-type="fig" rid="fig1s1">Figure 1—figure supplement 1</xref>)</title><p>In addition to the correlation analysis described above, Fisher’s exact test and the Mann-Whitney test were used to assess the correlation between RRM domains, RS repeats, and phase separation. Because in each of these analyses, two factors were being compared against a third factor, Bonferroni’s adjustment was used to set the significance threshold to p&lt;0.025. For the cases in which Fisher’s exact test were used, contingency tables are included in Supplementary Files (<xref ref-type="supplementary-material" rid="supp3 supp4">Supplementary files 3-4</xref>).</p><p>Because the number of RS repeats and percent R/S composition both occur across a broad distribution, the Mann-Whitney test was employed to determine to what extent these distributions differed between proteins in condensates and proteins outside of condensates. Because RRM domains are associated with both phase separation and an increase in the number of RS repeats, these variables were separated. The Mann-Whitney test is suitable for non-normal distributions with different sample sizes (<xref ref-type="bibr" rid="bib73">Widen et al., 2020</xref>). It is applicable in cases in which the sample size is greater than 30. It should be noted that the sample size of our smallest category in <xref ref-type="fig" rid="fig1">Figure 1B</xref> and <xref ref-type="fig" rid="fig1s1">Figure 1—figure supplement 1B</xref> (Proteins that have RRM domains but do not phase separate).</p></sec><sec id="s4-11"><title>Imaging</title><p>An SRSF1 construct (SRSF1 C16S/C148S/N220C) was tagged with Maleimide-Alexa488, dissolved into 800 mM Arg/Glu, pH 6.5, 0.2 mM TCEP, and stored at a concentration of 28 µM. Surplus Alexa488 dye was removed by a desalting column. The protein was then diluted to its final concentration in 100 mM KCl, 10 mM MES pH 6, 0.1 mM TCEP, with or without short peptides in a 96-well Cellvis glass bottom plate coated with Pluronics F127. Images are brightfield/GFP channel overlays taken on a Cytation5 imager using the software Gen5 3.10. More than three biological replicates were performed for phase separation experiments.</p></sec></sec></body><back><sec sec-type="additional-information" id="s5"><title>Additional information</title><fn-group content-type="competing-interest"><title>Competing interests</title><fn fn-type="COI-statement" id="conf1"><p>No competing interests declared</p></fn></fn-group><fn-group content-type="author-contribution"><title>Author contributions</title><fn fn-type="con" id="con1"><p>Data curation, Software, Formal analysis, Investigation, Visualization, Methodology, Writing - original draft, Writing - review and editing</p></fn><fn fn-type="con" id="con2"><p>Data curation, Formal analysis</p></fn><fn fn-type="con" id="con3"><p>Data curation, Formal analysis</p></fn><fn fn-type="con" id="con4"><p>Investigation</p></fn><fn fn-type="con" id="con5"><p>Investigation</p></fn><fn fn-type="con" id="con6"><p>Investigation</p></fn><fn fn-type="con" id="con7"><p>Investigation</p></fn><fn fn-type="con" id="con8"><p>Conceptualization, Data curation, Software, Formal analysis, Supervision, Funding acquisition, Validation, Investigation, Methodology, Writing - original draft, Project administration, Writing - review and editing</p></fn></fn-group></sec><sec sec-type="supplementary-material" id="s6"><title>Additional files</title><supplementary-material id="supp1"><label>Supplementary file 1.</label><caption><title>Effect of increasing length on phase separation for all possible dipeptide repeat combinations.</title><p>Lengths (x) analyzed are x=0, x&gt;2, x&gt;4, x&gt;6, and x&gt;8. <italic>p-</italic>values were obtained using a two-tailed correlation analysis. A p-value of &lt;0.05 was considered an indicator that increasing repeat length correlated significantly with fraction of proteins found in condensates. A population-based error of <inline-formula><mml:math id="inf2"><mml:mfrac><mml:mrow><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:msqrt><mml:msub><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>p</mml:mi><mml:mi>s</mml:mi></mml:mrow></mml:msub></mml:msqrt></mml:mrow></mml:mfrac></mml:math></inline-formula> was used to identify whether the sample size was large enough to draw conclusions from the data (as discussed in more detail in the methods section). Whether a repeat type passed both <italic>p</italic>-value and population-based error criteria is indicated in the right-most column.</p></caption><media xlink:href="elife-84412-supp1-v2.xlsx" mimetype="application" mime-subtype="xlsx"/></supplementary-material><supplementary-material id="supp2"><label>Supplementary file 2.</label><caption><title>Analysis of the effect of increasing repeat length when datasets are size-matched to either n=14 (sheet 1) or n=6 (sheet 2).</title><p>Percentage of proteins in condensates was found through Python’s random selection tool 50 different times for each population. Average and standard deviation are indicated at the top of the sheet.</p></caption><media xlink:href="elife-84412-supp2-v2.xlsx" mimetype="application" mime-subtype="xlsx"/></supplementary-material><supplementary-material id="supp3"><label>Supplementary file 3.</label><caption><title>Contingency tables analyzing the likelihood of RRM domain and RS repat co-occurrence both for proteins in condensates (left) and proteins not in condensates (right).</title><p>Values used correspond to those displayed in <xref ref-type="fig" rid="fig1">Figure 1A</xref>. <italic>p</italic>-values were obtained using Fisher’s exact test.</p></caption><media xlink:href="elife-84412-supp3-v2.xlsx" mimetype="application" mime-subtype="xlsx"/></supplementary-material><supplementary-material id="supp4"><label>Supplementary file 4.</label><caption><title>Contingency tables corresponding to the <italic>p</italic>-values shown in <xref ref-type="fig" rid="fig1">Figure 1C</xref> and <xref ref-type="fig" rid="fig1s1">Figure 1—figure supplement 1C</xref>.</title><p>Correlation between the fraction of proteins in condensates and the presence of a threshold number of short repeats or percentage R/S composition. <italic>p</italic>-values were obtained using Fisher’s exact test.</p></caption><media xlink:href="elife-84412-supp4-v2.xlsx" mimetype="application" mime-subtype="xlsx"/></supplementary-material><supplementary-material id="supp5"><label>Supplementary file 5.</label><caption><title>Values corresponding to the distance between peptides and hotspot residues at various timepoints in the MD simulation.</title><p>Distances (r) correspond to the length between the cysteine to which the paramagnetic center is attached and the NH hydrogen on the backbone of each residue under observation. Distances were measured using the Pymol measurement feature. Residue 42 is near to two peptides. Therefore, two distances are provided for residue 42. Bleached residues are expected to be within 12–15 Å of the paramagnetic center (<xref ref-type="bibr" rid="bib36">Iwahara et al., 2007</xref>).</p></caption><media xlink:href="elife-84412-supp5-v2.xlsx" mimetype="application" mime-subtype="xlsx"/></supplementary-material><supplementary-material id="mdar"><label>MDAR checklist</label><media xlink:href="elife-84412-mdarchecklist1-v2.docx" mimetype="application" mime-subtype="docx"/></supplementary-material></sec><sec sec-type="data-availability" id="s7"><title>Data availability</title><p>NMR assignment has been deposited to BMRB (ID: 51299).</p><p>The following dataset was generated:</p><p><element-citation publication-type="data" specific-use="isSupplementedBy" id="dataset1"><person-group person-group-type="author"><name><surname>Fargason</surname><given-names>T</given-names></name><name><surname>Zhang</surname><given-names>J</given-names></name></person-group><year iso-8601-date="2022">2022</year><data-title>NMR assignment for SRSF1</data-title><source>Biological Magnetic Resonance Data Bank</source><pub-id pub-id-type="accession" xlink:href="https://bmrb.io/data_library/summary/?bmrbId=51299">51299</pub-id></element-citation></p></sec><ack id="ack"><title>Acknowledgements</title><p>We want to thank UAB Central Alabama High-Field NMR Facility. We also want to acknowledge Dr. Jinfa Ying in Ad Bax lab in at NIDDK, Dr. Charles D Schwieters at NIH for technical support. This work is supported by U.S. National Science Foundation, MCB and U.S. National Institutes of Health, NIGMS. This work was supported by the U.S National Science Foundation [MCB2024964 to JZ] and U.S National Institutes of Health [R35GM147091-01 to JZ]. Funding for open access charge: National Science Foundation and National Institutes of Health.</p></ack><ref-list><title>References</title><ref id="bib1"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ahn</surname><given-names>EY</given-names></name><name><surname>DeKelver</surname><given-names>RC</given-names></name><name><surname>Lo</surname><given-names>MC</given-names></name><name><surname>Nguyen</surname><given-names>TA</given-names></name><name><surname>Matsuura</surname><given-names>S</given-names></name><name><surname>Boyapati</surname><given-names>A</given-names></name><name><surname>Pandit</surname><given-names>S</given-names></name><name><surname>Fu</surname><given-names>XD</given-names></name><name><surname>Zhang</surname><given-names>DE</given-names></name></person-group><year iso-8601-date="2011">2011</year><article-title>Son controls cell-cycle progression by coordinated regulation of RNA splicing</article-title><source>Molecular Cell</source><volume>42</volume><fpage>185</fpage><lpage>198</lpage><pub-id pub-id-type="doi">10.1016/j.molcel.2011.03.014</pub-id><pub-id pub-id-type="pmid">21504830</pub-id></element-citation></ref><ref id="bib2"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Aubol</surname><given-names>BE</given-names></name><name><surname>Plocinik</surname><given-names>RM</given-names></name><name><surname>Hagopian</surname><given-names>JC</given-names></name><name><surname>Ma</surname><given-names>CT</given-names></name><name><surname>McGlone</surname><given-names>ML</given-names></name><name><surname>Bandyopadhyay</surname><given-names>R</given-names></name><name><surname>Fu</surname><given-names>XD</given-names></name><name><surname>Adams</surname><given-names>JA</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>Partitioning RS domain phosphorylation in an SR protein through the CLK and SRPK protein kinases</article-title><source>Journal of Molecular Biology</source><volume>425</volume><fpage>2894</fpage><lpage>2909</lpage><pub-id pub-id-type="doi">10.1016/j.jmb.2013.05.013</pub-id><pub-id pub-id-type="pmid">23707382</pub-id></element-citation></ref><ref id="bib3"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Aubol</surname><given-names>BE</given-names></name><name><surname>Keshwani</surname><given-names>MM</given-names></name><name><surname>Fattet</surname><given-names>L</given-names></name><name><surname>Adams</surname><given-names>JA</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Mobilization of a splicing factor through a nuclear kinase-kinase complex</article-title><source>The Biochemical Journal</source><volume>475</volume><fpage>677</fpage><lpage>690</lpage><pub-id pub-id-type="doi">10.1042/BCJ20170672</pub-id><pub-id pub-id-type="pmid">29335301</pub-id></element-citation></ref><ref id="bib4"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Azpurua</surname><given-names>J</given-names></name><name><surname>El-Karim</surname><given-names>EG</given-names></name><name><surname>Tranquille</surname><given-names>M</given-names></name><name><surname>Dubnau</surname><given-names>J</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>A behavioral screen for mediators of age-dependent TDP-43 neurodegeneration identifies SF2/SRSF1 among A group of potent suppressors in both neurons and glia</article-title><source>PLOS Genetics</source><volume>17</volume><elocation-id>e1009882</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pgen.1009882</pub-id><pub-id pub-id-type="pmid">34723963</pub-id></element-citation></ref><ref id="bib5"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Banani</surname><given-names>SF</given-names></name><name><surname>Lee</surname><given-names>HO</given-names></name><name><surname>Hyman</surname><given-names>AA</given-names></name><name><surname>Rosen</surname><given-names>MK</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>Biomolecular condensates: organizers of cellular biochemistry</article-title><source>Nature Reviews. Molecular Cell Biology</source><volume>18</volume><fpage>285</fpage><lpage>298</lpage><pub-id pub-id-type="doi">10.1038/nrm.2017.7</pub-id><pub-id pub-id-type="pmid">28225081</pub-id></element-citation></ref><ref id="bib6"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Banjade</surname><given-names>S</given-names></name><name><surname>Rosen</surname><given-names>MK</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>Phase transitions of multivalent proteins can promote clustering of membrane receptors</article-title><source>eLife</source><volume>3</volume><elocation-id>e04123</elocation-id><pub-id pub-id-type="doi">10.7554/eLife.04123</pub-id><pub-id pub-id-type="pmid">25321392</pub-id></element-citation></ref><ref id="bib7"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Blencowe</surname><given-names>BJ</given-names></name><name><surname>Bowman</surname><given-names>JA</given-names></name><name><surname>McCracken</surname><given-names>S</given-names></name><name><surname>Rosonina</surname><given-names>E</given-names></name></person-group><year iso-8601-date="1999">1999</year><article-title>Sr-Related proteins and the processing of messenger RNA precursors</article-title><source>Biochemistry and Cell Biology = Biochimie et Biologie Cellulaire</source><volume>77</volume><fpage>277</fpage><lpage>291</lpage><pub-id pub-id-type="pmid">10546891</pub-id></element-citation></ref><ref id="bib8"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Boucher</surname><given-names>L</given-names></name><name><surname>Ouzounis</surname><given-names>CA</given-names></name><name><surname>Enright</surname><given-names>AJ</given-names></name><name><surname>Blencowe</surname><given-names>BJ</given-names></name></person-group><year iso-8601-date="2001">2001</year><article-title>A genome-wide survey of RS domain proteins</article-title><source>RNA</source><volume>7</volume><fpage>1693</fpage><lpage>1701</lpage><pub-id pub-id-type="pmid">11780626</pub-id></element-citation></ref><ref id="bib9"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Brangwynne</surname><given-names>CP</given-names></name><name><surname>Eckmann</surname><given-names>CR</given-names></name><name><surname>Courson</surname><given-names>DS</given-names></name><name><surname>Rybarska</surname><given-names>A</given-names></name><name><surname>Hoege</surname><given-names>C</given-names></name><name><surname>Gharakhani</surname><given-names>J</given-names></name><name><surname>Jülicher</surname><given-names>F</given-names></name><name><surname>Hyman</surname><given-names>AA</given-names></name></person-group><year iso-8601-date="2009">2009</year><article-title>Germline P granules are liquid droplets that localize by controlled dissolution/condensation</article-title><source>Science</source><volume>324</volume><fpage>1729</fpage><lpage>1732</lpage><pub-id pub-id-type="doi">10.1126/science.1172046</pub-id><pub-id pub-id-type="pmid">19460965</pub-id></element-citation></ref><ref id="bib10"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Burgess</surname><given-names>RR</given-names></name></person-group><year iso-8601-date="2009">2009</year><article-title>Protein precipitation techniques</article-title><source>Methods in Enzymology</source><volume>463</volume><fpage>331</fpage><lpage>342</lpage><pub-id pub-id-type="doi">10.1016/S0076-6879(09)63020-2</pub-id><pub-id pub-id-type="pmid">19892180</pub-id></element-citation></ref><ref id="bib11"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Burke</surname><given-names>KA</given-names></name><name><surname>Janke</surname><given-names>AM</given-names></name><name><surname>Rhine</surname><given-names>CL</given-names></name><name><surname>Fawzi</surname><given-names>NL</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>Residue-by-residue view of in vitro FUS granules that bind the C-terminal domain of RNA polymerase II</article-title><source>Molecular Cell</source><volume>60</volume><fpage>231</fpage><lpage>241</lpage><pub-id pub-id-type="doi">10.1016/j.molcel.2015.09.006</pub-id><pub-id pub-id-type="pmid">26455390</pub-id></element-citation></ref><ref id="bib12"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Cascarina</surname><given-names>SM</given-names></name><name><surname>King</surname><given-names>DC</given-names></name><name><surname>Osborne Nishimura</surname><given-names>E</given-names></name><name><surname>Ross</surname><given-names>ED</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>LCD-composer: an intuitive, composition-centric method enabling the identification and detailed functional mapping of low-complexity domains</article-title><source>NAR Genomics and Bioinformatics</source><volume>3</volume><elocation-id>lqab048</elocation-id><pub-id pub-id-type="doi">10.1093/nargab/lqab048</pub-id><pub-id pub-id-type="pmid">34056598</pub-id></element-citation></ref><ref id="bib13"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Cascarina</surname><given-names>SM</given-names></name><name><surname>Ross</surname><given-names>ED</given-names></name></person-group><year iso-8601-date="2022">2022</year><article-title>Expansion and functional analysis of the SR-related protein family across the domains of life</article-title><source>RNA</source><volume>28</volume><fpage>1298</fpage><lpage>1314</lpage><pub-id pub-id-type="doi">10.1261/rna.079170.122</pub-id><pub-id pub-id-type="pmid">35863866</pub-id></element-citation></ref><ref id="bib14"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Cho</surname><given-names>S</given-names></name><name><surname>Hoang</surname><given-names>A</given-names></name><name><surname>Sinha</surname><given-names>R</given-names></name><name><surname>Zhong</surname><given-names>XY</given-names></name><name><surname>Fu</surname><given-names>XD</given-names></name><name><surname>Krainer</surname><given-names>AR</given-names></name><name><surname>Ghosh</surname><given-names>G</given-names></name></person-group><year iso-8601-date="2011">2011</year><article-title>Interaction between the RNA binding domains of ser-arg splicing factor 1 and U1-70K snrnp protein determines early spliceosome assembly</article-title><source>PNAS</source><volume>108</volume><fpage>8233</fpage><lpage>8238</lpage><pub-id pub-id-type="doi">10.1073/pnas.1017700108</pub-id><pub-id pub-id-type="pmid">21536904</pub-id></element-citation></ref><ref id="bib15"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Cléry</surname><given-names>A</given-names></name><name><surname>Sinha</surname><given-names>R</given-names></name><name><surname>Anczuków</surname><given-names>O</given-names></name><name><surname>Corrionero</surname><given-names>A</given-names></name><name><surname>Moursy</surname><given-names>A</given-names></name><name><surname>Daubner</surname><given-names>GM</given-names></name><name><surname>Valcárcel</surname><given-names>J</given-names></name><name><surname>Krainer</surname><given-names>AR</given-names></name><name><surname>Allain</surname><given-names>FHT</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>Isolated pseudo-RNA-recognition motifs of SR proteins can regulate splicing using a noncanonical mode of RNA recognition</article-title><source>PNAS</source><volume>110</volume><fpage>E2802</fpage><lpage>E2811</lpage><pub-id pub-id-type="doi">10.1073/pnas.1303445110</pub-id><pub-id pub-id-type="pmid">23836656</pub-id></element-citation></ref><ref id="bib16"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Cléry</surname><given-names>A</given-names></name><name><surname>Krepl</surname><given-names>M</given-names></name><name><surname>Nguyen</surname><given-names>CKX</given-names></name><name><surname>Moursy</surname><given-names>A</given-names></name><name><surname>Jorjani</surname><given-names>H</given-names></name><name><surname>Katsantoni</surname><given-names>M</given-names></name><name><surname>Okoniewski</surname><given-names>M</given-names></name><name><surname>Mittal</surname><given-names>N</given-names></name><name><surname>Zavolan</surname><given-names>M</given-names></name><name><surname>Sponer</surname><given-names>J</given-names></name><name><surname>Allain</surname><given-names>FHT</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>Structure of SRSF1 RRM1 bound to RNA reveals an unexpected bimodal mode of interaction and explains its involvement in SMN1 exon7 splicing</article-title><source>Nature Communications</source><volume>12</volume><elocation-id>428</elocation-id><pub-id pub-id-type="doi">10.1038/s41467-020-20481-w</pub-id><pub-id pub-id-type="pmid">33462199</pub-id></element-citation></ref><ref id="bib17"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Clore</surname><given-names>GM</given-names></name><name><surname>Iwahara</surname><given-names>J</given-names></name></person-group><year iso-8601-date="2009">2009</year><article-title>Theory, practice, and applications of paramagnetic relaxation enhancement for the characterization of transient low-population states of biological macromolecules and their complexes</article-title><source>Chemical Reviews</source><volume>109</volume><fpage>4108</fpage><lpage>4139</lpage><pub-id pub-id-type="doi">10.1021/cr900033p</pub-id><pub-id pub-id-type="pmid">19522502</pub-id></element-citation></ref><ref id="bib18"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Delaglio</surname><given-names>F</given-names></name><name><surname>Grzesiek</surname><given-names>S</given-names></name><name><surname>Vuister</surname><given-names>GW</given-names></name><name><surname>Zhu</surname><given-names>G</given-names></name><name><surname>Pfeifer</surname><given-names>J</given-names></name><name><surname>Bax</surname><given-names>A</given-names></name></person-group><year iso-8601-date="1995">1995</year><article-title>NMRPipe: a multidimensional spectral processing system based on UNIX pipes</article-title><source>Journal of Biomolecular NMR</source><volume>6</volume><fpage>277</fpage><lpage>293</lpage><pub-id pub-id-type="doi">10.1007/BF00197809</pub-id><pub-id pub-id-type="pmid">8520220</pub-id></element-citation></ref><ref id="bib19"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Dundr</surname><given-names>M</given-names></name><name><surname>Misteli</surname><given-names>T</given-names></name></person-group><year iso-8601-date="2010">2010</year><article-title>Biogenesis of nuclear bodies</article-title><source>Cold Spring Harbor Perspectives in Biology</source><volume>2</volume><elocation-id>a000711</elocation-id><pub-id pub-id-type="doi">10.1101/cshperspect.a000711</pub-id><pub-id pub-id-type="pmid">21068152</pub-id></element-citation></ref><ref id="bib20"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Emmanouilidis</surname><given-names>L</given-names></name><name><surname>Esteban-Hofer</surname><given-names>L</given-names></name><name><surname>Jeschke</surname><given-names>G</given-names></name><name><surname>Allain</surname><given-names>FH-T</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>Structural biology of RNA-binding proteins in the context of phase separation: what NMR and EPR can bring?</article-title><source>Current Opinion in Structural Biology</source><volume>70</volume><fpage>132</fpage><lpage>138</lpage><pub-id pub-id-type="doi">10.1016/j.sbi.2021.07.001</pub-id><pub-id pub-id-type="pmid">34371262</pub-id></element-citation></ref><ref id="bib21"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Fargason</surname><given-names>T</given-names></name><name><surname>Wang</surname><given-names>T</given-names></name><name><surname>De Silva</surname><given-names>NIU</given-names></name><name><surname>Zhang</surname><given-names>Z</given-names></name><name><surname>McKelvey</surname><given-names>H</given-names></name><name><surname>Knapp</surname><given-names>T</given-names></name><name><surname>Zaharias</surname><given-names>S</given-names></name><name><surname>Zhang</surname><given-names>J</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>Amide additives improve RDC measurements in polyacrylamide</article-title><source>Journal of Biomolecular NMR</source><volume>74</volume><fpage>119</fpage><lpage>124</lpage><pub-id pub-id-type="doi">10.1007/s10858-020-00305-1</pub-id><pub-id pub-id-type="pmid">32056065</pub-id></element-citation></ref><ref id="bib22"><element-citation publication-type="software"><person-group person-group-type="author"><name><surname>Fargason</surname><given-names>T</given-names></name></person-group><year iso-8601-date="2023">2023</year><data-title>Taliafargason / repeats_in_condensates</data-title><version designator="8d3d1b6">8d3d1b6</version><source>Github</source><ext-link ext-link-type="uri" xlink:href="https://github.com/taliafargason/Repeats_in_Condensates">https://github.com/taliafargason/Repeats_in_Condensates</ext-link></element-citation></ref><ref id="bib23"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Fei</surname><given-names>J</given-names></name><name><surname>Jadaliha</surname><given-names>M</given-names></name><name><surname>Harmon</surname><given-names>TS</given-names></name><name><surname>Li</surname><given-names>ITS</given-names></name><name><surname>Hua</surname><given-names>B</given-names></name><name><surname>Hao</surname><given-names>Q</given-names></name><name><surname>Holehouse</surname><given-names>AS</given-names></name><name><surname>Reyer</surname><given-names>M</given-names></name><name><surname>Sun</surname><given-names>Q</given-names></name><name><surname>Freier</surname><given-names>SM</given-names></name><name><surname>Pappu</surname><given-names>RV</given-names></name><name><surname>Prasanth</surname><given-names>KV</given-names></name><name><surname>Ha</surname><given-names>T</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>Quantitative analysis of multilayer organization of proteins and RNA in nuclear speckles at super resolution</article-title><source>Journal of Cell Science</source><volume>130</volume><fpage>4180</fpage><lpage>4192</lpage><pub-id pub-id-type="doi">10.1242/jcs.206854</pub-id><pub-id pub-id-type="pmid">29133588</pub-id></element-citation></ref><ref id="bib24"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Feng</surname><given-names>Y</given-names></name><name><surname>Bao</surname><given-names>W</given-names></name><name><surname>Luo</surname><given-names>Y</given-names></name><name><surname>Tian</surname><given-names>L</given-names></name><name><surname>Chen</surname><given-names>X</given-names></name><name><surname>Yi</surname><given-names>M</given-names></name><name><surname>Xiong</surname><given-names>H</given-names></name><name><surname>Huang</surname><given-names>Q</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>Phosphomimetic mutants of pigment epithelium-derived factor with enhanced anti-choroidal melanoma cell activity in vitro and in vivo</article-title><source>Investigative Opthalmology &amp; Visual Science</source><volume>53</volume><elocation-id>6793</elocation-id><pub-id pub-id-type="doi">10.1167/iovs.12-10326</pub-id></element-citation></ref><ref id="bib25"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ge</surname><given-names>H</given-names></name><name><surname>Manley</surname><given-names>JL</given-names></name></person-group><year iso-8601-date="1990">1990</year><article-title>A protein factor, ASF, controls cell-specific alternative splicing of SV40 early pre-mRNA in vitro</article-title><source>Cell</source><volume>62</volume><fpage>25</fpage><lpage>34</lpage><pub-id pub-id-type="doi">10.1016/0092-8674(90)90236-8</pub-id><pub-id pub-id-type="pmid">2163768</pub-id></element-citation></ref><ref id="bib26"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Golovanov</surname><given-names>AP</given-names></name><name><surname>Hautbergue</surname><given-names>GM</given-names></name><name><surname>Wilson</surname><given-names>SA</given-names></name><name><surname>Lian</surname><given-names>LY</given-names></name></person-group><year iso-8601-date="2004">2004</year><article-title>A simple method for improving protein solubility and long-term stability</article-title><source>Journal of the American Chemical Society</source><volume>126</volume><fpage>8933</fpage><lpage>8939</lpage><pub-id pub-id-type="doi">10.1021/ja049297h</pub-id><pub-id pub-id-type="pmid">15264823</pub-id></element-citation></ref><ref id="bib27"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Greenfield</surname><given-names>NJ</given-names></name></person-group><year iso-8601-date="2006">2006</year><article-title>Using circular dichroism spectra to estimate protein secondary structure</article-title><source>Nature Protocols</source><volume>1</volume><fpage>2876</fpage><lpage>2890</lpage><pub-id pub-id-type="doi">10.1038/nprot.2006.202</pub-id><pub-id pub-id-type="pmid">17406547</pub-id></element-citation></ref><ref id="bib28"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Greig</surname><given-names>JA</given-names></name><name><surname>Nguyen</surname><given-names>TA</given-names></name><name><surname>Lee</surname><given-names>M</given-names></name><name><surname>Holehouse</surname><given-names>AS</given-names></name><name><surname>Posey</surname><given-names>AE</given-names></name><name><surname>Pappu</surname><given-names>RV</given-names></name><name><surname>Jedd</surname><given-names>G</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>Arginine-enriched mixed-charge domains provide cohesion for nuclear speckle condensation</article-title><source>Molecular Cell</source><volume>77</volume><fpage>1237</fpage><lpage>1250</lpage><pub-id pub-id-type="doi">10.1016/j.molcel.2020.01.025</pub-id><pub-id pub-id-type="pmid">32048997</pub-id></element-citation></ref><ref id="bib29"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Gui</surname><given-names>JF</given-names></name><name><surname>Lane</surname><given-names>WS</given-names></name><name><surname>Fu</surname><given-names>XD</given-names></name></person-group><year iso-8601-date="1994">1994</year><article-title>A serine kinase regulates intracellular localization of splicing factors in the cell cycle</article-title><source>Nature</source><volume>369</volume><fpage>678</fpage><lpage>682</lpage><pub-id pub-id-type="doi">10.1038/369678a0</pub-id><pub-id pub-id-type="pmid">8208298</pub-id></element-citation></ref><ref id="bib30"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hamelberg</surname><given-names>D</given-names></name><name><surname>Shen</surname><given-names>T</given-names></name><name><surname>McCammon</surname><given-names>JA</given-names></name></person-group><year iso-8601-date="2007">2007</year><article-title>A proposed signaling motif for nuclear import in mrna processing via the formation of arginine claw</article-title><source>PNAS</source><volume>104</volume><fpage>14947</fpage><lpage>14951</lpage><pub-id pub-id-type="doi">10.1073/pnas.0703151104</pub-id><pub-id pub-id-type="pmid">17823247</pub-id></element-citation></ref><ref id="bib31"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hammarskjold</surname><given-names>ML</given-names></name><name><surname>Rekosh</surname><given-names>D</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>Sr proteins: to shuttle or not to shuttle, that is the question</article-title><source>The Journal of Cell Biology</source><volume>216</volume><fpage>1875</fpage><lpage>1877</lpage><pub-id pub-id-type="doi">10.1083/jcb.201705009</pub-id><pub-id pub-id-type="pmid">28600433</pub-id></element-citation></ref><ref id="bib32"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Handwerger</surname><given-names>KE</given-names></name><name><surname>Cordero</surname><given-names>JA</given-names></name><name><surname>Gall</surname><given-names>JG</given-names></name></person-group><year iso-8601-date="2005">2005</year><article-title>Cajal bodies, nucleoli, and speckles in the <italic>Xenopus</italic> oocyte nucleus have a low-density, sponge-like structure</article-title><source>Molecular Biology of the Cell</source><volume>16</volume><fpage>202</fpage><lpage>211</lpage><pub-id pub-id-type="doi">10.1091/mbc.e04-08-0742</pub-id><pub-id pub-id-type="pmid">15509651</pub-id></element-citation></ref><ref id="bib33"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Haward</surname><given-names>F</given-names></name><name><surname>Maslon</surname><given-names>MM</given-names></name><name><surname>Yeyati</surname><given-names>PL</given-names></name><name><surname>Bellora</surname><given-names>N</given-names></name><name><surname>Hansen</surname><given-names>JN</given-names></name><name><surname>Aitken</surname><given-names>S</given-names></name><name><surname>Lawson</surname><given-names>J</given-names></name><name><surname>von Kriegsheim</surname><given-names>A</given-names></name><name><surname>Wachten</surname><given-names>D</given-names></name><name><surname>Mill</surname><given-names>P</given-names></name><name><surname>Adams</surname><given-names>IR</given-names></name><name><surname>Caceres</surname><given-names>JF</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>Nucleo-Cytoplasmic shuttling of splicing factor SRSF1 is required for development and cilia function</article-title><source>eLife</source><volume>10</volume><elocation-id>e65104</elocation-id><pub-id pub-id-type="doi">10.7554/eLife.65104</pub-id><pub-id pub-id-type="pmid">34338635</pub-id></element-citation></ref><ref id="bib34"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ilik</surname><given-names>İA</given-names></name><name><surname>Malszycki</surname><given-names>M</given-names></name><name><surname>Lübke</surname><given-names>AK</given-names></name><name><surname>Schade</surname><given-names>C</given-names></name><name><surname>Meierhofer</surname><given-names>D</given-names></name><name><surname>Aktaş</surname><given-names>T</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>Son and SRRM2 are essential for nuclear speckle formation</article-title><source>eLife</source><volume>9</volume><elocation-id>e60579</elocation-id><pub-id pub-id-type="doi">10.7554/eLife.60579</pub-id><pub-id pub-id-type="pmid">33095160</pub-id></element-citation></ref><ref id="bib35"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Iwahara</surname><given-names>J</given-names></name><name><surname>Schwieters</surname><given-names>CD</given-names></name><name><surname>Clore</surname><given-names>GM</given-names></name></person-group><year iso-8601-date="2004">2004</year><article-title>Ensemble approach for NMR structure refinement against (1) H paramagnetic relaxation enhancement data arising from a flexible paramagnetic group attached to a macromolecule</article-title><source>Journal of the American Chemical Society</source><volume>126</volume><fpage>5879</fpage><lpage>5896</lpage><pub-id pub-id-type="doi">10.1021/ja031580d</pub-id><pub-id pub-id-type="pmid">15125681</pub-id></element-citation></ref><ref id="bib36"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Iwahara</surname><given-names>J</given-names></name><name><surname>Tang</surname><given-names>C</given-names></name><name><surname>Marius Clore</surname><given-names>G</given-names></name></person-group><year iso-8601-date="2007">2007</year><article-title>Practical aspects of 1H transverse paramagnetic relaxation enhancement measurements on macromolecules</article-title><source>Journal of Magnetic Resonance</source><volume>184</volume><fpage>185</fpage><lpage>195</lpage><pub-id pub-id-type="doi">10.1016/j.jmr.2006.10.003</pub-id><pub-id pub-id-type="pmid">17084097</pub-id></element-citation></ref><ref id="bib37"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Johnson</surname><given-names>BA</given-names></name></person-group><year iso-8601-date="2004">2004</year><article-title>Using nmrview to visualize and analyze the NMR spectra of macromolecules</article-title><source>Methods in Molecular Biology</source><volume>278</volume><fpage>313</fpage><lpage>352</lpage><pub-id pub-id-type="doi">10.1385/1-59259-809-9:313</pub-id><pub-id pub-id-type="pmid">15318002</pub-id></element-citation></ref><ref id="bib38"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kelly</surname><given-names>AE</given-names></name><name><surname>Ou</surname><given-names>HD</given-names></name><name><surname>Withers</surname><given-names>R</given-names></name><name><surname>Dötsch</surname><given-names>V</given-names></name></person-group><year iso-8601-date="2002">2002</year><article-title>Low-conductivity buffers for high-sensitivity NMR measurements</article-title><source>Journal of the American Chemical Society</source><volume>124</volume><fpage>12013</fpage><lpage>12019</lpage><pub-id pub-id-type="doi">10.1021/ja026121b</pub-id></element-citation></ref><ref id="bib39"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kim</surname><given-names>TH</given-names></name><name><surname>Tsang</surname><given-names>B</given-names></name><name><surname>Vernon</surname><given-names>RM</given-names></name><name><surname>Sonenberg</surname><given-names>N</given-names></name><name><surname>Kay</surname><given-names>LE</given-names></name><name><surname>Forman-Kay</surname><given-names>JD</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Phospho-dependent phase separation of FMRP and CAPRIN1 recapitulates regulation of translation and deadenylation</article-title><source>Science</source><volume>365</volume><fpage>825</fpage><lpage>829</lpage><pub-id pub-id-type="doi">10.1126/science.aax4240</pub-id><pub-id pub-id-type="pmid">31439799</pub-id></element-citation></ref><ref id="bib40"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kohtz</surname><given-names>JD</given-names></name><name><surname>Jamison</surname><given-names>SF</given-names></name><name><surname>Will</surname><given-names>CL</given-names></name><name><surname>Zuo</surname><given-names>P</given-names></name><name><surname>Lührmann</surname><given-names>R</given-names></name><name><surname>Garcia-Blanco</surname><given-names>MA</given-names></name><name><surname>Manley</surname><given-names>JL</given-names></name></person-group><year iso-8601-date="1994">1994</year><article-title>Protein-Protein interactions and 5’-splice-site recognition in mammalian mRNA precursors</article-title><source>Nature</source><volume>368</volume><fpage>119</fpage><lpage>124</lpage><pub-id pub-id-type="doi">10.1038/368119a0</pub-id><pub-id pub-id-type="pmid">8139654</pub-id></element-citation></ref><ref id="bib41"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Krainer</surname><given-names>AR</given-names></name><name><surname>Conway</surname><given-names>GC</given-names></name><name><surname>Kozak</surname><given-names>D</given-names></name></person-group><year iso-8601-date="1990">1990a</year><article-title>The essential pre-mrna splicing factor SF2 influences 5’ splice site selection by activating proximal sites</article-title><source>Cell</source><volume>62</volume><fpage>35</fpage><lpage>42</lpage><pub-id pub-id-type="doi">10.1016/0092-8674(90)90237-9</pub-id><pub-id pub-id-type="pmid">2364434</pub-id></element-citation></ref><ref id="bib42"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Krainer</surname><given-names>AR</given-names></name><name><surname>Conway</surname><given-names>GC</given-names></name><name><surname>Kozak</surname><given-names>D</given-names></name></person-group><year iso-8601-date="1990">1990b</year><article-title>Purification and characterization of pre-mrna splicing factor SF2 from hela cells</article-title><source>Genes &amp; Development</source><volume>4</volume><fpage>1158</fpage><lpage>1171</lpage><pub-id pub-id-type="doi">10.1101/gad.4.7.1158</pub-id><pub-id pub-id-type="pmid">2145194</pub-id></element-citation></ref><ref id="bib43"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Lafontaine</surname><given-names>DLJ</given-names></name><name><surname>Riback</surname><given-names>JA</given-names></name><name><surname>Bascetin</surname><given-names>R</given-names></name><name><surname>Brangwynne</surname><given-names>CP</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>The nucleolus as a multiphase liquid condensate</article-title><source>Nature Reviews Molecular Cell Biology</source><volume>22</volume><fpage>165</fpage><lpage>182</lpage><pub-id pub-id-type="doi">10.1038/s41580-020-0272-6</pub-id><pub-id pub-id-type="pmid">32873929</pub-id></element-citation></ref><ref id="bib44"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Lamanna</surname><given-names>AC</given-names></name><name><surname>Karbstein</surname><given-names>K</given-names></name></person-group><year iso-8601-date="2009">2009</year><article-title>Nob1 binds the single-stranded cleavage site D at the 3’-end of 18S rrna with its PIN domain</article-title><source>PNAS</source><volume>106</volume><fpage>14259</fpage><lpage>14264</lpage><pub-id pub-id-type="doi">10.1073/pnas.0905403106</pub-id><pub-id pub-id-type="pmid">19706509</pub-id></element-citation></ref><ref id="bib45"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Lamond</surname><given-names>AI</given-names></name><name><surname>Spector</surname><given-names>DL</given-names></name></person-group><year iso-8601-date="2003">2003</year><article-title>Nuclear speckles: a model for nuclear organelles</article-title><source>Nature Reviews. Molecular Cell Biology</source><volume>4</volume><fpage>605</fpage><lpage>612</lpage><pub-id pub-id-type="doi">10.1038/nrm1172</pub-id><pub-id pub-id-type="pmid">12923522</pub-id></element-citation></ref><ref id="bib46"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Larkin</surname><given-names>MA</given-names></name><name><surname>Blackshields</surname><given-names>G</given-names></name><name><surname>Brown</surname><given-names>NP</given-names></name><name><surname>Chenna</surname><given-names>R</given-names></name><name><surname>McGettigan</surname><given-names>PA</given-names></name><name><surname>McWilliam</surname><given-names>H</given-names></name><name><surname>Valentin</surname><given-names>F</given-names></name><name><surname>Wallace</surname><given-names>IM</given-names></name><name><surname>Wilm</surname><given-names>A</given-names></name><name><surname>Lopez</surname><given-names>R</given-names></name><name><surname>Thompson</surname><given-names>JD</given-names></name><name><surname>Gibson</surname><given-names>TJ</given-names></name><name><surname>Higgins</surname><given-names>DG</given-names></name></person-group><year iso-8601-date="2007">2007</year><article-title>Clustal W and Clustal X version 2.0</article-title><source>Bioinformatics</source><volume>23</volume><fpage>2947</fpage><lpage>2948</lpage><pub-id pub-id-type="doi">10.1093/bioinformatics/btm404</pub-id><pub-id pub-id-type="pmid">17846036</pub-id></element-citation></ref><ref id="bib47"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Li</surname><given-names>P</given-names></name><name><surname>Banjade</surname><given-names>S</given-names></name><name><surname>Cheng</surname><given-names>H-C</given-names></name><name><surname>Kim</surname><given-names>S</given-names></name><name><surname>Chen</surname><given-names>B</given-names></name><name><surname>Guo</surname><given-names>L</given-names></name><name><surname>Llaguno</surname><given-names>M</given-names></name><name><surname>Hollingsworth</surname><given-names>JV</given-names></name><name><surname>King</surname><given-names>DS</given-names></name><name><surname>Banani</surname><given-names>SF</given-names></name><name><surname>Russo</surname><given-names>PS</given-names></name><name><surname>Jiang</surname><given-names>Q-X</given-names></name><name><surname>Nixon</surname><given-names>BT</given-names></name><name><surname>Rosen</surname><given-names>MK</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>Phase transitions in the assembly of multivalent signalling proteins</article-title><source>Nature</source><volume>483</volume><fpage>336</fpage><lpage>340</lpage><pub-id pub-id-type="doi">10.1038/nature10879</pub-id><pub-id pub-id-type="pmid">22398450</pub-id></element-citation></ref><ref id="bib48"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Li</surname><given-names>Q</given-names></name><name><surname>Peng</surname><given-names>X</given-names></name><name><surname>Li</surname><given-names>Y</given-names></name><name><surname>Tang</surname><given-names>W</given-names></name><name><surname>Zhu</surname><given-names>J</given-names></name><name><surname>Huang</surname><given-names>J</given-names></name><name><surname>Qi</surname><given-names>Y</given-names></name><name><surname>Zhang</surname><given-names>Z</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>LLPSDB: a database of proteins undergoing liquid-liquid phase separation in vitro</article-title><source>Nucleic Acids Research</source><volume>48</volume><fpage>D320</fpage><lpage>D327</lpage><pub-id pub-id-type="doi">10.1093/nar/gkz778</pub-id><pub-id pub-id-type="pmid">31906602</pub-id></element-citation></ref><ref id="bib49"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Li</surname><given-names>K</given-names></name><name><surname>Wang</surname><given-names>Z</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>Speckles and paraspeckles coordinate to regulate HSV-1 genes transcription</article-title><source>Communications Biology</source><volume>4</volume><elocation-id>1207</elocation-id><pub-id pub-id-type="doi">10.1038/s42003-021-02742-6</pub-id><pub-id pub-id-type="pmid">34675360</pub-id></element-citation></ref><ref id="bib50"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Manley</surname><given-names>JL</given-names></name><name><surname>Krainer</surname><given-names>AR</given-names></name></person-group><year iso-8601-date="2010">2010</year><article-title>A rational Nomenclature for serine/arginine-rich protein splicing factors (SR proteins)</article-title><source>Genes &amp; Development</source><volume>24</volume><fpage>1073</fpage><lpage>1074</lpage><pub-id pub-id-type="doi">10.1101/gad.1934910</pub-id><pub-id pub-id-type="pmid">20516191</pub-id></element-citation></ref><ref id="bib51"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Martin</surname><given-names>EW</given-names></name><name><surname>Thomasen</surname><given-names>FE</given-names></name><name><surname>Milkovic</surname><given-names>NM</given-names></name><name><surname>Cuneo</surname><given-names>MJ</given-names></name><name><surname>Grace</surname><given-names>CR</given-names></name><name><surname>Nourse</surname><given-names>A</given-names></name><name><surname>Lindorff-Larsen</surname><given-names>K</given-names></name><name><surname>Mittag</surname><given-names>T</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>Interplay of folded domains and the disordered low-complexity domain in mediating hnrnpa1 phase separation</article-title><source>Nucleic Acids Research</source><volume>49</volume><fpage>2931</fpage><lpage>2945</lpage><pub-id pub-id-type="doi">10.1093/nar/gkab063</pub-id><pub-id pub-id-type="pmid">33577679</pub-id></element-citation></ref><ref id="bib52"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Molliex</surname><given-names>A</given-names></name><name><surname>Temirov</surname><given-names>J</given-names></name><name><surname>Lee</surname><given-names>J</given-names></name><name><surname>Coughlin</surname><given-names>M</given-names></name><name><surname>Kanagaraj</surname><given-names>AP</given-names></name><name><surname>Kim</surname><given-names>HJ</given-names></name><name><surname>Mittag</surname><given-names>T</given-names></name><name><surname>Taylor</surname><given-names>JP</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>Phase separation by low complexity domains promotes stress granule assembly and drives pathological fibrillization</article-title><source>Cell</source><volume>163</volume><fpage>123</fpage><lpage>133</lpage><pub-id pub-id-type="doi">10.1016/j.cell.2015.09.015</pub-id><pub-id pub-id-type="pmid">26406374</pub-id></element-citation></ref><ref id="bib53"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Monahan</surname><given-names>Z</given-names></name><name><surname>Ryan</surname><given-names>VH</given-names></name><name><surname>Janke</surname><given-names>AM</given-names></name><name><surname>Burke</surname><given-names>KA</given-names></name><name><surname>Rhoads</surname><given-names>SN</given-names></name><name><surname>Zerze</surname><given-names>GH</given-names></name><name><surname>O’Meally</surname><given-names>R</given-names></name><name><surname>Dignon</surname><given-names>GL</given-names></name><name><surname>Conicella</surname><given-names>AE</given-names></name><name><surname>Zheng</surname><given-names>W</given-names></name><name><surname>Best</surname><given-names>RB</given-names></name><name><surname>Cole</surname><given-names>RN</given-names></name><name><surname>Mittal</surname><given-names>J</given-names></name><name><surname>Shewmaker</surname><given-names>F</given-names></name><name><surname>Fawzi</surname><given-names>NL</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>Phosphorylation of the FUS low‐complexity domain disrupts phase separation, aggregation, and toxicity</article-title><source>The EMBO Journal</source><volume>36</volume><fpage>2951</fpage><lpage>2967</lpage><pub-id pub-id-type="doi">10.15252/embj.201696394</pub-id><pub-id pub-id-type="pmid">28790177</pub-id></element-citation></ref><ref id="bib54"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Murthy</surname><given-names>AC</given-names></name><name><surname>Dignon</surname><given-names>GL</given-names></name><name><surname>Kan</surname><given-names>Y</given-names></name><name><surname>Zerze</surname><given-names>GH</given-names></name><name><surname>Parekh</surname><given-names>SH</given-names></name><name><surname>Mittal</surname><given-names>J</given-names></name><name><surname>Fawzi</surname><given-names>NL</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Molecular interactions underlying liquid-liquid phase separation of the FUS low-complexity domain</article-title><source>Nature Structural &amp; Molecular Biology</source><volume>26</volume><fpage>637</fpage><lpage>648</lpage><pub-id pub-id-type="doi">10.1038/s41594-019-0250-x</pub-id><pub-id pub-id-type="pmid">31270472</pub-id></element-citation></ref><ref id="bib55"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Murthy</surname><given-names>AC</given-names></name><name><surname>Fawzi</surname><given-names>NL</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>The (un) structural biology of biomolecular liquid-liquid phase separation using NMR spectroscopy</article-title><source>The Journal of Biological Chemistry</source><volume>295</volume><fpage>2375</fpage><lpage>2384</lpage><pub-id pub-id-type="doi">10.1074/jbc.REV119.009847</pub-id><pub-id pub-id-type="pmid">31911439</pub-id></element-citation></ref><ref id="bib56"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Neugebauer</surname><given-names>KM</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>Special focus on the Cajal body</article-title><source>RNA Biology</source><volume>14</volume><fpage>669</fpage><lpage>670</lpage><pub-id pub-id-type="doi">10.1080/15476286.2017.1316928</pub-id><pub-id pub-id-type="pmid">28486008</pub-id></element-citation></ref><ref id="bib57"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ngo</surname><given-names>JCK</given-names></name><name><surname>Giang</surname><given-names>K</given-names></name><name><surname>Chakrabarti</surname><given-names>S</given-names></name><name><surname>Ma</surname><given-names>C-T</given-names></name><name><surname>Huynh</surname><given-names>N</given-names></name><name><surname>Hagopian</surname><given-names>JC</given-names></name><name><surname>Dorrestein</surname><given-names>PC</given-names></name><name><surname>Fu</surname><given-names>X-D</given-names></name><name><surname>Adams</surname><given-names>JA</given-names></name><name><surname>Ghosh</surname><given-names>G</given-names></name></person-group><year iso-8601-date="2008">2008</year><article-title>A sliding docking interaction is essential for sequential and processive phosphorylation of an SR protein by SRPK1</article-title><source>Molecular Cell</source><volume>29</volume><fpage>563</fpage><lpage>576</lpage><pub-id pub-id-type="doi">10.1016/j.molcel.2007.12.017</pub-id><pub-id pub-id-type="pmid">18342604</pub-id></element-citation></ref><ref id="bib58"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ning</surname><given-names>W</given-names></name><name><surname>Guo</surname><given-names>Y</given-names></name><name><surname>Lin</surname><given-names>S</given-names></name><name><surname>Mei</surname><given-names>B</given-names></name><name><surname>Wu</surname><given-names>Y</given-names></name><name><surname>Jiang</surname><given-names>P</given-names></name><name><surname>Tan</surname><given-names>X</given-names></name><name><surname>Zhang</surname><given-names>W</given-names></name><name><surname>Chen</surname><given-names>G</given-names></name><name><surname>Peng</surname><given-names>D</given-names></name><name><surname>Chu</surname><given-names>L</given-names></name><name><surname>Xue</surname><given-names>Y</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>DrLLPS: a data resource of liquid-liquid phase separation in eukaryotes</article-title><source>Nucleic Acids Research</source><volume>48</volume><fpage>D288</fpage><lpage>D295</lpage><pub-id pub-id-type="doi">10.1093/nar/gkz1027</pub-id><pub-id pub-id-type="pmid">31691822</pub-id></element-citation></ref><ref id="bib59"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Okuno</surname><given-names>Y</given-names></name><name><surname>Yoo</surname><given-names>J</given-names></name><name><surname>Schwieters</surname><given-names>CD</given-names></name><name><surname>Best</surname><given-names>RB</given-names></name><name><surname>Chung</surname><given-names>HS</given-names></name><name><surname>Clore</surname><given-names>GM</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>Atomic view of cosolute-induced protein denaturation probed by NMR solvent paramagnetic relaxation enhancement</article-title><source>PNAS</source><volume>118</volume><elocation-id>e2112021118</elocation-id><pub-id pub-id-type="doi">10.1073/pnas.2112021118</pub-id><pub-id pub-id-type="pmid">34404723</pub-id></element-citation></ref><ref id="bib60"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Putnam</surname><given-names>CD</given-names></name><name><surname>Hammel</surname><given-names>M</given-names></name><name><surname>Hura</surname><given-names>GL</given-names></name><name><surname>Tainer</surname><given-names>JA</given-names></name></person-group><year iso-8601-date="2007">2007</year><article-title>X-Ray solution scattering (SAXS) combined with crystallography and computation: defining accurate macromolecular structures, conformations and assemblies in solution</article-title><source>Quarterly Reviews of Biophysics</source><volume>40</volume><fpage>191</fpage><lpage>285</lpage><pub-id pub-id-type="doi">10.1017/S0033583507004635</pub-id><pub-id pub-id-type="pmid">18078545</pub-id></element-citation></ref><ref id="bib61"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Reber</surname><given-names>S</given-names></name><name><surname>Jutzi</surname><given-names>D</given-names></name><name><surname>Lindsay</surname><given-names>H</given-names></name><name><surname>Devoy</surname><given-names>A</given-names></name><name><surname>Mechtersheimer</surname><given-names>J</given-names></name><name><surname>Levone</surname><given-names>BR</given-names></name><name><surname>Domanski</surname><given-names>M</given-names></name><name><surname>Bentmann</surname><given-names>E</given-names></name><name><surname>Dormann</surname><given-names>D</given-names></name><name><surname>Mühlemann</surname><given-names>O</given-names></name><name><surname>Barabino</surname><given-names>SML</given-names></name><name><surname>Ruepp</surname><given-names>M-D</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>The phase separation-dependent FUS interactome reveals nuclear and cytoplasmic function of liquid-liquid phase separation</article-title><source>Nucleic Acids Research</source><volume>49</volume><fpage>7713</fpage><lpage>7731</lpage><pub-id pub-id-type="doi">10.1093/nar/gkab582</pub-id><pub-id pub-id-type="pmid">34233002</pub-id></element-citation></ref><ref id="bib62"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ryan</surname><given-names>VH</given-names></name><name><surname>Dignon</surname><given-names>GL</given-names></name><name><surname>Zerze</surname><given-names>GH</given-names></name><name><surname>Chabata</surname><given-names>CV</given-names></name><name><surname>Silva</surname><given-names>R</given-names></name><name><surname>Conicella</surname><given-names>AE</given-names></name><name><surname>Amaya</surname><given-names>J</given-names></name><name><surname>Burke</surname><given-names>KA</given-names></name><name><surname>Mittal</surname><given-names>J</given-names></name><name><surname>Fawzi</surname><given-names>NL</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Mechanistic view of hnRNPA2 low-complexity domain structure, interactions, and phase separation altered by mutation and arginine methylation</article-title><source>Molecular Cell</source><volume>69</volume><fpage>465</fpage><lpage>479</lpage><pub-id pub-id-type="doi">10.1016/j.molcel.2017.12.022</pub-id><pub-id pub-id-type="pmid">29358076</pub-id></element-citation></ref><ref id="bib63"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Schwieters</surname><given-names>CD</given-names></name><name><surname>Kuszewski</surname><given-names>JJ</given-names></name><name><surname>Tjandra</surname><given-names>N</given-names></name><name><surname>Marius Clore</surname><given-names>G</given-names></name></person-group><year iso-8601-date="2003">2003</year><article-title>The xplor-NIH NMR molecular structure determination package</article-title><source>Journal of Magnetic Resonance</source><volume>160</volume><fpage>65</fpage><lpage>73</lpage><pub-id pub-id-type="doi">10.1016/S1090-7807(02)00014-9</pub-id><pub-id pub-id-type="pmid">12565051</pub-id></element-citation></ref><ref id="bib64"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Schwieters</surname><given-names>C</given-names></name><name><surname>Kuszewski</surname><given-names>J</given-names></name><name><surname>Mariusclore</surname><given-names>G</given-names></name></person-group><year iso-8601-date="2006">2006</year><article-title>Using xplor–NIH for NMR molecular structure determination</article-title><source>Progress in Nuclear Magnetic Resonance Spectroscopy</source><volume>48</volume><fpage>47</fpage><lpage>62</lpage><pub-id pub-id-type="doi">10.1016/j.pnmrs.2005.10.001</pub-id></element-citation></ref><ref id="bib65"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Screaton</surname><given-names>GR</given-names></name><name><surname>Cáceres</surname><given-names>JF</given-names></name><name><surname>Mayeda</surname><given-names>A</given-names></name><name><surname>Bell</surname><given-names>MV</given-names></name><name><surname>Plebanski</surname><given-names>M</given-names></name><name><surname>Jackson</surname><given-names>DG</given-names></name><name><surname>Bell</surname><given-names>JI</given-names></name><name><surname>Krainer</surname><given-names>AR</given-names></name></person-group><year iso-8601-date="1995">1995</year><article-title>Identification and characterization of three members of the human SR family of pre-mRNA splicing factors</article-title><source>The EMBO Journal</source><volume>14</volume><fpage>4336</fpage><lpage>4349</lpage><pub-id pub-id-type="doi">10.1002/j.1460-2075.1995.tb00108.x</pub-id><pub-id pub-id-type="pmid">7556075</pub-id></element-citation></ref><ref id="bib66"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Shepard</surname><given-names>PJ</given-names></name><name><surname>Hertel</surname><given-names>KJ</given-names></name></person-group><year iso-8601-date="2009">2009</year><article-title>The SR protein family</article-title><source>Genome Biology</source><volume>10</volume><elocation-id>242</elocation-id><pub-id pub-id-type="doi">10.1186/gb-2009-10-10-242</pub-id><pub-id pub-id-type="pmid">19857271</pub-id></element-citation></ref><ref id="bib67"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Souquere</surname><given-names>S</given-names></name><name><surname>Mollet</surname><given-names>S</given-names></name><name><surname>Kress</surname><given-names>M</given-names></name><name><surname>Dautry</surname><given-names>F</given-names></name><name><surname>Pierron</surname><given-names>G</given-names></name><name><surname>Weil</surname><given-names>D</given-names></name></person-group><year iso-8601-date="2009">2009</year><article-title>Unravelling the ultrastructure of stress granules and associated P-bodies in human cells</article-title><source>Journal of Cell Science</source><volume>122</volume><fpage>3619</fpage><lpage>3626</lpage><pub-id pub-id-type="doi">10.1242/jcs.054437</pub-id><pub-id pub-id-type="pmid">19812307</pub-id></element-citation></ref><ref id="bib68"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Strulson</surname><given-names>CA</given-names></name><name><surname>Molden</surname><given-names>RC</given-names></name><name><surname>Keating</surname><given-names>CD</given-names></name><name><surname>Bevilacqua</surname><given-names>PC</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>Rna catalysis through compartmentalization</article-title><source>Nature Chemistry</source><volume>4</volume><fpage>941</fpage><lpage>946</lpage><pub-id pub-id-type="doi">10.1038/nchem.1466</pub-id><pub-id pub-id-type="pmid">23089870</pub-id></element-citation></ref><ref id="bib69"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Tacke</surname><given-names>R</given-names></name><name><surname>Manley</surname><given-names>JL</given-names></name></person-group><year iso-8601-date="1995">1995</year><article-title>The human splicing factors ASF/SF2 and SC35 possess distinct, functionally significant RNA binding specificities</article-title><source>The EMBO Journal</source><volume>14</volume><fpage>3540</fpage><lpage>3551</lpage><pub-id pub-id-type="doi">10.1002/j.1460-2075.1995.tb07360.x</pub-id><pub-id pub-id-type="pmid">7543047</pub-id></element-citation></ref><ref id="bib70"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Trevino</surname><given-names>SR</given-names></name><name><surname>Scholtz</surname><given-names>JM</given-names></name><name><surname>Pace</surname><given-names>CN</given-names></name></person-group><year iso-8601-date="2008">2008</year><article-title>Measuring and increasing protein solubility</article-title><source>Journal of Pharmaceutical Sciences</source><volume>97</volume><fpage>4155</fpage><lpage>4166</lpage><pub-id pub-id-type="doi">10.1002/jps.21327</pub-id></element-citation></ref><ref id="bib71"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Tripathi</surname><given-names>V</given-names></name><name><surname>Song</surname><given-names>DY</given-names></name><name><surname>Zong</surname><given-names>X</given-names></name><name><surname>Shevtsov</surname><given-names>SP</given-names></name><name><surname>Hearn</surname><given-names>S</given-names></name><name><surname>Fu</surname><given-names>XD</given-names></name><name><surname>Dundr</surname><given-names>M</given-names></name><name><surname>Prasanth</surname><given-names>KV</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>Srsf1 regulates the assembly of pre-mRNA processing factors in nuclear speckles</article-title><source>Molecular Biology of the Cell</source><volume>23</volume><fpage>3694</fpage><lpage>3706</lpage><pub-id pub-id-type="doi">10.1091/mbc.E12-03-0206</pub-id><pub-id pub-id-type="pmid">22855529</pub-id></element-citation></ref><ref id="bib72"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname><given-names>A</given-names></name><name><surname>Conicella</surname><given-names>AE</given-names></name><name><surname>Schmidt</surname><given-names>HB</given-names></name><name><surname>Martin</surname><given-names>EW</given-names></name><name><surname>Rhoads</surname><given-names>SN</given-names></name><name><surname>Reeb</surname><given-names>AN</given-names></name><name><surname>Nourse</surname><given-names>A</given-names></name><name><surname>Ramirez Montero</surname><given-names>D</given-names></name><name><surname>Ryan</surname><given-names>VH</given-names></name><name><surname>Rohatgi</surname><given-names>R</given-names></name><name><surname>Shewmaker</surname><given-names>F</given-names></name><name><surname>Naik</surname><given-names>MT</given-names></name><name><surname>Mittag</surname><given-names>T</given-names></name><name><surname>Ayala</surname><given-names>YM</given-names></name><name><surname>Fawzi</surname><given-names>NL</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>A single N-terminal phosphomimic disrupts TDP-43 polymerization, phase separation, and RNA splicing</article-title><source>The EMBO Journal</source><volume>37</volume><elocation-id>e97452</elocation-id><pub-id pub-id-type="doi">10.15252/embj.201797452</pub-id><pub-id pub-id-type="pmid">29438978</pub-id></element-citation></ref><ref id="bib73"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Widen</surname><given-names>JC</given-names></name><name><surname>Tholen</surname><given-names>M</given-names></name><name><surname>Yim</surname><given-names>JJ</given-names></name><name><surname>Bogyo</surname><given-names>M</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>Methods for analysis of near-infrared (NIR) quenched-fluorescent contrast agents in mouse models of cancer</article-title><source>Methods in Enzymology</source><volume>639</volume><fpage>141</fpage><lpage>166</lpage><pub-id pub-id-type="doi">10.1016/bs.mie.2020.04.012</pub-id><pub-id pub-id-type="pmid">32475399</pub-id></element-citation></ref><ref id="bib74"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Wider</surname><given-names>G</given-names></name><name><surname>Dreier</surname><given-names>L</given-names></name></person-group><year iso-8601-date="2006">2006</year><article-title>Measuring protein concentrations by NMR spectroscopy</article-title><source>Journal of the American Chemical Society</source><volume>128</volume><fpage>2571</fpage><lpage>2576</lpage><pub-id pub-id-type="doi">10.1021/ja055336t</pub-id></element-citation></ref><ref id="bib75"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Wong</surname><given-names>LE</given-names></name><name><surname>Kim</surname><given-names>TH</given-names></name><name><surname>Muhandiram</surname><given-names>DR</given-names></name><name><surname>Forman-Kay</surname><given-names>JD</given-names></name><name><surname>Kay</surname><given-names>LE</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>Nmr experiments for studies of dilute and condensed protein phases: application to the phase-separating protein CAPRIN1</article-title><source>Journal of the American Chemical Society</source><volume>142</volume><fpage>2471</fpage><lpage>2489</lpage><pub-id pub-id-type="doi">10.1021/jacs.9b12208</pub-id><pub-id pub-id-type="pmid">31898464</pub-id></element-citation></ref><ref id="bib76"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Xiang</surname><given-names>S</given-names></name><name><surname>Gapsys</surname><given-names>V</given-names></name><name><surname>Kim</surname><given-names>HY</given-names></name><name><surname>Bessonov</surname><given-names>S</given-names></name><name><surname>Hsiao</surname><given-names>HH</given-names></name><name><surname>Möhlmann</surname><given-names>S</given-names></name><name><surname>Klaukien</surname><given-names>V</given-names></name><name><surname>Ficner</surname><given-names>R</given-names></name><name><surname>Becker</surname><given-names>S</given-names></name><name><surname>Urlaub</surname><given-names>H</given-names></name><name><surname>Lührmann</surname><given-names>R</given-names></name><name><surname>de Groot</surname><given-names>B</given-names></name><name><surname>Zweckstetter</surname><given-names>M</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>Phosphorylation drives a dynamic switch in serine/arginine-rich proteins</article-title><source>Structure</source><volume>21</volume><fpage>2162</fpage><lpage>2174</lpage><pub-id pub-id-type="doi">10.1016/j.str.2013.09.014</pub-id><pub-id pub-id-type="pmid">24183573</pub-id></element-citation></ref><ref id="bib77"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Xu</surname><given-names>S</given-names></name><name><surname>Lai</surname><given-names>SK</given-names></name><name><surname>Sim</surname><given-names>DY</given-names></name><name><surname>Ang</surname><given-names>WSL</given-names></name><name><surname>Li</surname><given-names>HY</given-names></name><name><surname>Roca</surname><given-names>X</given-names></name></person-group><year iso-8601-date="2022">2022</year><article-title>SRRM2 organizes splicing condensates to regulate alternative splicing</article-title><source>Nucleic Acids Research</source><volume>50</volume><fpage>8599</fpage><lpage>8614</lpage><pub-id pub-id-type="doi">10.1093/nar/gkac669</pub-id><pub-id pub-id-type="pmid">35929045</pub-id></element-citation></ref><ref id="bib78"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Yang</surname><given-names>Z</given-names></name><name><surname>Jakymiw</surname><given-names>A</given-names></name><name><surname>Wood</surname><given-names>MR</given-names></name><name><surname>Eystathioy</surname><given-names>T</given-names></name><name><surname>Rubin</surname><given-names>RL</given-names></name><name><surname>Fritzler</surname><given-names>MJ</given-names></name><name><surname>Chan</surname><given-names>EKL</given-names></name></person-group><year iso-8601-date="2004">2004</year><article-title>Gw182 is critical for the stability of GW bodies expressed during the cell cycle and cell proliferation</article-title><source>Journal of Cell Science</source><volume>117</volume><fpage>5567</fpage><lpage>5578</lpage><pub-id pub-id-type="doi">10.1242/jcs.01477</pub-id><pub-id pub-id-type="pmid">15494374</pub-id></element-citation></ref><ref id="bib79"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>You</surname><given-names>K</given-names></name><name><surname>Huang</surname><given-names>Q</given-names></name><name><surname>Yu</surname><given-names>C</given-names></name><name><surname>Shen</surname><given-names>B</given-names></name><name><surname>Sevilla</surname><given-names>C</given-names></name><name><surname>Shi</surname><given-names>M</given-names></name><name><surname>Hermjakob</surname><given-names>H</given-names></name><name><surname>Chen</surname><given-names>Y</given-names></name><name><surname>Li</surname><given-names>T</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>PhaSepDB: a database of liquid-liquid phase separation related proteins</article-title><source>Nucleic Acids Research</source><volume>48</volume><fpage>D354</fpage><lpage>D359</lpage><pub-id pub-id-type="doi">10.1093/nar/gkz847</pub-id><pub-id pub-id-type="pmid">31584089</pub-id></element-citation></ref><ref id="bib80"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname><given-names>J</given-names></name><name><surname>McCann</surname><given-names>KL</given-names></name><name><surname>Qiu</surname><given-names>C</given-names></name><name><surname>Gonzalez</surname><given-names>LE</given-names></name><name><surname>Baserga</surname><given-names>SJ</given-names></name><name><surname>Hall</surname><given-names>TMT</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Nop9 is a PUF-like protein that prevents premature cleavage to correctly process pre-18S rrna</article-title><source>Nature Communications</source><volume>7</volume><elocation-id>13085</elocation-id><pub-id pub-id-type="doi">10.1038/ncomms13085</pub-id><pub-id pub-id-type="pmid">27725644</pub-id></element-citation></ref></ref-list></back><sub-article article-type="editor-report" id="sa0"><front-stub><article-id pub-id-type="doi">10.7554/eLife.84412.sa0</article-id><title-group><article-title>Editor's evaluation</article-title></title-group><contrib-group><contrib contrib-type="author"><name><surname>Black</surname><given-names>Douglas L</given-names></name><role specific-use="editor">Reviewing Editor</role><aff><institution-wrap><institution-id institution-id-type="ror">https://ror.org/046rm7j60</institution-id><institution>University of California, Los Angeles</institution></institution-wrap><country>United States</country></aff></contrib></contrib-group><related-object id="sa0ro1" object-id-type="id" object-id="10.1101/2022.10.24.511151" link-type="continued-by" xlink:href="https://sciety.org/articles/activity/10.1101/2022.10.24.511151"/></front-stub><body><p>This study convincingly demonstrates that the splicing factor SRSF1 can be solubilized in the presence of short RS or ER containing peptides, and uses this discovery to determine the solution NMR structure of SRSF1, as well as to map its interactions with RS peptides. These findings are important in that SR proteins are key regulators of alternative splicing but their study has been greatly hampered by their low solubility. The development of a general method that allows their structural and biochemical analysis in solution will have broad applications.</p></body></sub-article><sub-article article-type="decision-letter" id="sa1"><front-stub><article-id pub-id-type="doi">10.7554/eLife.84412.sa1</article-id><title-group><article-title>Decision letter</article-title></title-group><contrib-group content-type="section"><contrib contrib-type="editor"><name><surname>Black</surname><given-names>Douglas L</given-names></name><role>Reviewing Editor</role><aff><institution-wrap><institution-id institution-id-type="ror">https://ror.org/046rm7j60</institution-id><institution>University of California, Los Angeles</institution></institution-wrap><country>United States</country></aff></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name><surname>Kojetin</surname><given-names>Douglas J</given-names></name><role>Reviewer</role><aff><institution>UF Scripps Biomedical Research</institution><country>United States</country></aff></contrib></contrib-group></front-stub><body><boxed-text id="sa2-box1"><p>Our editorial process produces two outputs: (i) <ext-link ext-link-type="uri" xlink:href="https://sciety.org/articles/activity/10.1101/2022.10.24.511151">public reviews</ext-link> designed to be posted alongside <ext-link ext-link-type="uri" xlink:href="https://www.biorxiv.org/content/10.1101/2022.10.24.511151v2">the preprint</ext-link> for the benefit of readers; (ii) feedback on the manuscript for the authors, including requests for revisions, shown below. We also include an acceptance summary that explains what the editors found interesting or important about the work.</p></boxed-text><p><bold>Decision letter after peer review:</bold></p><p>Thank you for submitting your article &quot;Peptides that Mimic RS repeats modulate phase separation of SRSF1, revealing a reliance on combined stacking and electrostatic interactions&quot; for consideration by <italic>eLife</italic>. Your article has been reviewed by 2 peer reviewers, and the evaluation has been overseen by Douglas Black as the Reviewing Editor and James Manley as the Senior Editor. The following individual involved in the review of your submission has agreed to reveal their identity: Douglas J Kojetin (Reviewer #2).</p><p>The reviewers have discussed their reviews with one another, and the Reviewing Editor has drafted this to help you prepare a revised submission. This paper addresses a major limitation in the study of SRSF1, a critical and well-known alternative splicing factor. SRSF1 and other members of this family exhibit limited solubility due to their tendency to undergo liquid phase separation driven by their RS domains. The authors tested the hypothesis that short RS repeat peptides might compete with the intermolecular interactions that drive phase separation, and allow sufficient solubilization for biochemical and structural studies. The data convincingly show that short peptides with RS, ER, or DR repeats can render recombinant SRSF1 soluble. They go on to analyze the newly concentrated protein by NMR and present well-resolved and assignable NMR spectra of SRSF1 dissolved with RS8 as a co-solute. The authors use paramagnetic relaxation enhancement experiments to map interactions between RS8 and SRSF1. Their findings suggest that the peptide interactions with the RNA binding domains (RRMs) are similar to those of the RS domain and are mediated by a combination of ionic interactions between arginine and acidic side chains, and pi-stacking interactions with surface-exposed hydrophobic residues.</p><p>A second part of the study seeks to identify features of proteins that make them more likely to undergo phase separations. The authors use a bioinformatics approach to correlate the presence of RS repeats with the identity of proteins in databases of phase-separated material. They also use molecular modeling and software tools to predict additional RRM domains that might be prone to phase separation by looking for those containing acidic amino acids with neighboring surface-exposed aromatic/hydrophobic residues. These findings are much less convincing than the structural and biochemical studies. The findings for SR proteins, which localize to nuclear speckles are extrapolated to all kinds of RNA-binding proteins and cellular condensates. The observed fold enrichments are small. The sample sizes decrease rapidly with increasing repeat length, and it is not clear if the appropriate statistical tests were performed to correct for multiple hypothesis testing. Importantly, these predictions derived from the computational analyses are not tested experimentally.</p><p>Overall, the reviewers found the NMR and PRE data to be convincing. The use of short RS, ER, and DR peptides as co-solutes for insoluble SR-domain proteins is a novel and clever advance. The computational analyses are incomplete and over-interpreted. These findings would require experimental validation.</p><p>The reviewers and editor all found the solubilization and structural data presented in Figures 1 to 6A to be a valuable advance. However, the computational studies of Figure 6 B – F were found to be too preliminary and need validation to add much value to the manuscript. It is recommended that the authors remove Figure 6 B – F, or provide experimental validation for their interpretations of these findings. The paper should be revised to emphasize Figures 1 – 6A, to address the following essential points.</p><p>Essential revisions:</p><p>1. In Figure 1A, the sample sizes of proteins containing no RS, 2RS, 4RS etc. vary over a wide range. The authors should address whether the observed differences between percentages remain significant if size-matched subsets of the larger pools are sampled at random (multiple times). These will provide a more appropriate comparison set for the longer repeat-containing proteins. Please explain why the analysis was restricted to short RS repeats, and how short RS repeat proteins (2-8AA) can be considered representative of the SR protein family.</p><p>2. In Figure 1A, the authors should also assess other dipeptides and their repeats. It seems reasonably straightforward to repeat this analysis with every possible dipeptide to assess whether there is something special about SR, or if similar enrichments are seen with other dipeptide combinations, perhaps derived from other types of low-complexity domains.</p><p>3. In panel 1B, where multiple populations are compared against each other, have the p-values been corrected for multiple hypothesis testing? This criticism pertains to all of the informatics analyses throughout the manuscript. Additional details of how the statistical analyses were performed need to be explained in the methods.</p><p>4. For the PRE NMR-based structure calculations of the RS8/SRSF1(dRS) interactions using Xplor-NIH, additional information and analyses would improve the rigor of the reported studies. How many total independent Xplor-NIH structures were calculated? Was a single Xplor-NIH calculated structure used for MD simulations? How many independent MD simulations were performed? Was a single RS8 peptide included in the MD simulations? At what frequency are the interactions shown in Figure 5 populated in each simulation – could this be shown via distance histograms (https://amberhub.chpc.utah.edu/generating-histograms-with-cpptraj/)? Were there any restraints (PRE?) used for the MD simulations; if not, are the MD results consistent with the experimental PRE NMR data? The methods section does not describe what programs were used for MD trajectory analysis.</p><p><italic>Reviewer #1 (Recommendations for the authors):</italic></p><p>The strategy of using short RS (and similar peptides) to solubilize SFRS1 is clever, the experimental work is well done, and the NMR / PRE data are convincing. I have no major problems with this aspect of the work. The manuscript is well-written and easy to understand, with a few exceptions described below. I am less convinced by the conclusions drawn from the bioinformatic analyses. My primary concerns are detailed below.</p><p>1. In figure 1A, the sample size between no RS, 2RS, 4RS, … varies by quite a bit. Would the difference between percentages remain significant if sample size-matched subsets of the larger pools were selected at random (multiple times) for use as a comparison set with the longer repeat-containing proteins? Why was the analysis restricted to short RS repeats? How representative are short RS repeats (2-8AA) of the SR protein family?</p><p>2. In figure 1A, would you see a similar trend if you used a different dipeptide and repeats thereof? It seems reasonably straightforward to repeat this analysis with EVERY possible dipeptide to see if there is something special about SR, or if similar enrichment is seen with other dipeptide combinations.</p><p>3. In panel 1B, where multiple populations are compared against each other, have the p-values been corrected for multiple hypothesis testing? This criticism pertains to all of the informatics analyses throughout the manuscript. I looked for additional details concerning how the statistical analyses were performed in the methods, but could not find them.</p><p>4. In figure 6, it would be useful to repeat the analysis with just solved RRM structures, and just modeled RRM structures, and compare to the pooled data shown in Figure 6. Here again, it would be useful to know whether the p-values corrected for multiple hypothesis testing.</p><p>5. It should be relatively straightforward to test the model presented in the discussion, i.e. that acidic residues with neighboring hydrophobic residues are predictive of RRM proteins that localize to nuclear speckles. A mutational analysis demonstrating a change in speckle localization in cells, or even solubility in vitro, would help to test this model's validity.</p><p><italic>Reviewer #2 (Recommendations for the authors):</italic></p><p>Results section, 1st paragraph: The annotation of proteins that are present in condensates in the phase separation databases likely represents the minimal numbers as there are likely proteins not yet reported that can form or go into condensates. It would be useful to include a sentence in the Results section to indicate how this may over/underestimate the analysis in Figure 1.</p><p>Results section, 2nd subsection: Is there a published paper (that could be cited) that may have also inspired this thought? – &quot;The high Arg composition in these proteins inspired us to use high concentrations of Arg amino acid in our protocol to purify and solubilize SRSF1.&quot; There are several published papers indicating that adding Arg to buffers can enhance general solubility.</p><p>For the PRE NMR-based structure calculations of the RS8/SRSF1(dRS) interaction using Xplor-NIH, additional information and analyses could improve the scientific rigor of the reported work. How many total independent Xplor-NIH structures were calculated? Was a single Xplor-NIH calculated structure used for MD simulations? How many independent MD simulations were performed? Was a single RS8 peptide included in the MD simulations? At what frequency are the interactions shown in Figure 5 populated in each simulation-could this be shown via distance histograms (https://amberhub.chpc.utah.edu/generating-histograms-with-cpptraj/)? Were there any restraints (PRE?) used for the MD simulations; if not, are the MD results consistent with the experimental PRE NMR data? The methods section does not describe what program(s) was(were) used for MD trajectory analysis.</p><p>Although five potential solubilizing repeat peptides were tested in Figures 2 and 3, it is possible that further enhancements to solubility could be gained from a larger structure-activity relationship (SAR) analysis. Perhaps the authors could allude to this in the manuscript (discussion?).</p></body></sub-article><sub-article article-type="reply" id="sa2"><front-stub><article-id pub-id-type="doi">10.7554/eLife.84412.sa2</article-id><title-group><article-title>Author response</article-title></title-group></front-stub><body><disp-quote content-type="editor-comment"><p>Essential revisions:</p><p>1. In Figure 1A, the sample sizes of proteins containing no RS, 2RS, 4RS etc. vary over a wide range. The authors should address whether the observed differences between percentages remain significant if size-matched subsets of the larger pools are sampled at random (multiple times). These will provide a more appropriate comparison set for the longer repeat-containing proteins. Please explain why the analysis was restricted to short RS repeats, and how short RS repeat proteins (2-8AA) can be considered representative of the SR protein family.</p></disp-quote><p>As suggested, we did 50 random samplings using subsets of 6 and 14 proteins (Table S2). From these two sets of simulations, we observed similar percentage differences for proteins with different RS lengths. Therefore, the difference we observed is not due to sample size.</p><p>Our work is focused on RS-containing proteins, which include both SR proteins and SR-related proteins. The first SR protein, SRSF1, was discovered by Dr. Manley and Dr. Krainer over 30 years ago. The field uses their definition of the SR protein family. The SR protein family was originally defined as: any protein that has one or two N-terminal RRMs, followed by a downstream RS domain of at least 50 amino acids with &gt; 40% RS content, characterized by consecutive RS or SR repeats. In addition to domain and amino acid composition, biological functions and posttranslational modifications are also considerations in defining SR proteins. Based on these criteria, twelve proteins are classified as SR proteins. The proteins that have RS repeats but do not meet all criteria are classified as SR-related or SR-like proteins.</p><p>It is noteworthy that the definition of SR-related proteins is ambiguous. Proposed standards are either based on percent composition(1) or number of 2mer and 4mer repeats(2), not a threshold length. Most SR or SR-like proteins have interrupted RS repeats. For this reason, the population of proteins decreases dramatically along with increased RS length. The longest uninterrupted RS region (16 aa, 8 dipeptide repeats) is found in SRSF1. Therefore, we searched for short RS repeats (2-8aa). We noticed similar trends when we searched by percent enrichment rather than repeat number, which we have added to our supplemental. A discussion of this has been added to the Results section.</p><disp-quote content-type="editor-comment"><p>2. In Figure 1A, the authors should also assess other dipeptides and their repeats. It seems reasonably straightforward to repeat this analysis with every possible dipeptide to assess whether there is something special about SR, or if similar enrichments are seen with other dipeptide combinations, perhaps derived from other types of low-complexity domains.</p></disp-quote><p>We appreciate this insightful comment. As suggested, we analyzed how phase separation tendency is correlated with repeat length for all dipeptide motifs. In our analysis, we assumed that the two amino acids are interchangeable. For example, RS and SR motifs are considered the same motif. In total, we analyzed 210 dipeptide motifs (190 motifs with two different amino acids, and 20 motifs with the same amino acid). Two criteria were used to select dipeptide motifs whose length is reliably correlated to phase separation. The first criterion is that the Pearson correlation p-value is smaller than 0.05. The second one is that error of the phase separation probability is smaller than half of the fraction of proteins in condensates. The error was calculated as <inline-formula><mml:math id="sa2m1"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mn>1</mml:mn><mml:mrow><mml:mo>/</mml:mo></mml:mrow><mml:msqrt><mml:msub><mml:mi>N</mml:mi><mml:mrow><mml:mtext>ps</mml:mtext></mml:mrow></mml:msub></mml:msqrt></mml:mrow></mml:mstyle></mml:math></inline-formula>, where <italic>N<sub>ps</sub></italic> is the number of proteins with 8-mers found condensates. This estimation resembles the margin of error. The second criterion is related to the number of proteins and therefore reflects the stability of the data point. This criterion filters out the repeats that have only a few proteins (usually 1 or 2). In total, 6 motifs pass these criteria: GG, KK, QQ, PP, RG, and RS. Of these dipeptide repeats, proteins regions rich in Gly, Gln, Pro, RG/RGG, and RS have been reported to drive phase separation. It is noteworthy that 8-mer RS-containing proteins and 6-mer RS-containing proteins have the highest tendency to phase separate among all of these motifs.</p><disp-quote content-type="editor-comment"><p>3. In panel 1B, where multiple populations are compared against each other, have the p-values been corrected for multiple hypothesis testing? This criticism pertains to all of the informatics analyses throughout the manuscript. Additional details of how the statistical analyses were performed need to be explained in the methods.</p></disp-quote><p>We appreciate this insight. In Figure 1B, we are testing the hypothesis that the number of RS repeats is correlated with a tendency to appear in condensates. However, RRM domains are also correlated with both phase separation and RS repeats. Because many RS-containing proteins have RRM domains, we separated these two variables.</p><p>You bring up a great point that this results in two factors (RRM, phase separation) being analyzed against RS repeat number. To eliminate the possibility that random chance produces a spuriously low <italic>p</italic>-value, we also performed Bonferroni’s adjustment for two hypothesis testing, which would make our significance threshold 0.025 rather than 0.05. In Figure 1B, we considered the difference between columns 1 and 2 (p &lt; 0.0001) and the difference between columns 5 and 6 (p = 0.0014) to be significant. In contrast, we found the difference between columns 3 and 4 (p = 0.139) and the difference between columns 7 and 8 (0.1449) to not be significant. These <italic>p</italic>-values were obtained using the Mann-Whitney test, which is suitable for a sample size &gt; 30 as well as a situation in which sample sizes are very different and distribution is not normal. We have revised the figure legend and methods section.</p><disp-quote content-type="editor-comment"><p>4. For the PRE NMR-based structure calculations of the RS8/SRSF1(dRS) interactions using Xplor-NIH, additional information and analyses would improve the rigor of the reported studies. How many total independent Xplor-NIH structures were calculated? Was a single Xplor-NIH calculated structure used for MD simulations? How many independent MD simulations were performed? Was a single RS8 peptide included in the MD simulations? At what frequency are the interactions shown in Figure 5 populated in each simulation – could this be shown via distance histograms (https://amberhub.chpc.utah.edu/generating-histograms-with-cpptraj/)? Were there any restraints (PRE?) used for the MD simulations; if not, are the MD results consistent with the experimental PRE NMR data? The methods section does not describe what programs were used for MD trajectory analysis.</p></disp-quote><p>We have updated our methods section and added additional information regarding population and PRE fitting to our supplemental (Figures S7 and S8). In total, 100 Xplor-NIH structures were calculated using the ensemble size of 10. In Xplor-NIH calculations, the RRM1/RRM2 linker and RS peptides were allowed full range of motion. RRM1 was held in place, and RRM2 was allowed to move as a group. Each Xplor-NIH structure had 1 peptide (resulting in 10 peptides total in the ensemble). The ensemble structure that best fit the PRE data had a Q factor of 0.419 and Pearson Correlation Coefficient of 0.938 Using the top Xplor NIH structures, we created a starting structure for MD simulation with 4 peptides (one for each of the major PRE hotspots). This structure did not fully account for all PRE values but met the distance expectations for bleached residues (&lt;12-15 Å). Two MD simulations are shown in this paper—one in which no restraints were applied and one in which peptide 2 was held in contact with residue F88 on the loop region following the β<sub>4</sub> strand. During the simulation, the nature of interactions within the hotspots changed in some cases, but the distance between peptides and bleached residues did not significantly change. The MD trajectory analysis was performed by CPPTRAJ.</p></body></sub-article></article>