<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Archiving and Interchange DTD with MathML3 v1.3 20210610//EN"  "JATS-archivearticle1-3-mathml3.dtd"><article xmlns:ali="http://www.niso.org/schemas/ali/1.0/" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article" dtd-version="1.3"><front><journal-meta><journal-id journal-id-type="nlm-ta">elife</journal-id><journal-id journal-id-type="publisher-id">eLife</journal-id><journal-title-group><journal-title>eLife</journal-title></journal-title-group><issn publication-format="electronic" pub-type="epub">2050-084X</issn><publisher><publisher-name>eLife Sciences Publications, Ltd</publisher-name></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">87335</article-id><article-id pub-id-type="doi">10.7554/eLife.87335</article-id><article-id pub-id-type="doi" specific-use="version">10.7554/eLife.87335.3</article-id><article-version article-version-type="publication-state">version of record</article-version><article-categories><subj-group subj-group-type="display-channel"><subject>Research Article</subject></subj-group><subj-group subj-group-type="heading"><subject>Evolutionary Biology</subject></subj-group><subj-group subj-group-type="heading"><subject>Genetics and Genomics</subject></subj-group></article-categories><title-group><article-title>The protein domains of vertebrate species in which selection is more effective have greater intrinsic structural disorder</article-title></title-group><contrib-group><contrib contrib-type="author" equal-contrib="yes"><name><surname>Weibel</surname><given-names>Catherine A</given-names></name><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0003-1837-5209</contrib-id><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="fn" rid="equal-contrib1">†</xref><xref ref-type="fn" rid="pa1">‡</xref><xref ref-type="other" rid="fund1"/><xref ref-type="other" rid="fund3"/><xref ref-type="other" rid="fund4"/><xref ref-type="fn" rid="con1"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author" equal-contrib="yes"><name><surname>Wheeler</surname><given-names>Andrew L</given-names></name><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0002-5347-5419</contrib-id><xref ref-type="aff" rid="aff3">3</xref><xref ref-type="fn" rid="equal-contrib1">†</xref><xref ref-type="other" rid="fund2"/><xref ref-type="fn" rid="con2"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author"><name><surname>James</surname><given-names>Jennifer E</given-names></name><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0003-0518-6783</contrib-id><xref ref-type="aff" rid="aff4">4</xref><xref ref-type="fn" rid="pa2">§</xref><xref ref-type="other" rid="fund1"/><xref ref-type="other" rid="fund3"/><xref ref-type="fn" rid="con3"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author"><name><surname>Willis</surname><given-names>Sara M</given-names></name><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0002-1605-6426</contrib-id><xref ref-type="aff" rid="aff4">4</xref><xref ref-type="fn" rid="pa3">#</xref><xref ref-type="other" rid="fund1"/><xref ref-type="other" rid="fund3"/><xref ref-type="fn" rid="con4"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author"><name><surname>McShea</surname><given-names>Hanon</given-names></name><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0002-9341-4899</contrib-id><xref ref-type="aff" rid="aff5">5</xref><xref ref-type="other" rid="fund7"/><xref ref-type="fn" rid="con5"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author" corresp="yes"><name><surname>Masel</surname><given-names>Joanna</given-names></name><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0002-7398-2127</contrib-id><email>masel@arizona.edu</email><xref ref-type="aff" rid="aff4">4</xref><xref ref-type="other" rid="fund1"/><xref ref-type="other" rid="fund3"/><xref ref-type="fn" rid="con6"/><xref ref-type="fn" rid="conf1"/></contrib><aff id="aff1"><label>1</label><institution-wrap><institution-id institution-id-type="ror">https://ror.org/03m2x1q45</institution-id><institution>Department of Mathematics, University of Arizona</institution></institution-wrap><addr-line><named-content content-type="city">Tucson</named-content></addr-line><country>United States</country></aff><aff id="aff2"><label>2</label><institution-wrap><institution-id institution-id-type="ror">https://ror.org/03m2x1q45</institution-id><institution>Department of Physics, University of Arizona</institution></institution-wrap><addr-line><named-content content-type="city">Tucson</named-content></addr-line><country>United States</country></aff><aff id="aff3"><label>3</label><institution-wrap><institution-id institution-id-type="ror">https://ror.org/03m2x1q45</institution-id><institution>Genetics Graduate Interdisciplinary Program, University of Arizona</institution></institution-wrap><addr-line><named-content content-type="city">Tucson</named-content></addr-line><country>United States</country></aff><aff id="aff4"><label>4</label><institution-wrap><institution-id institution-id-type="ror">https://ror.org/03m2x1q45</institution-id><institution>Department of Ecology and Evolutionary Biology, University of Arizona</institution></institution-wrap><addr-line><named-content content-type="city">Tucson</named-content></addr-line><country>United States</country></aff><aff id="aff5"><label>5</label><institution-wrap><institution-id institution-id-type="ror">https://ror.org/00f54p054</institution-id><institution>Department of Earth System Science, Stanford University</institution></institution-wrap><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib-group><contrib-group content-type="section"><contrib contrib-type="editor"><name><surname>Ogbunugafor</surname><given-names>C Brandon</given-names></name><role>Reviewing Editor</role><aff><institution-wrap><institution-id institution-id-type="ror">https://ror.org/03v76x132</institution-id><institution>Yale University</institution></institution-wrap><country>United States</country></aff></contrib><contrib contrib-type="senior_editor"><name><surname>Weigel</surname><given-names>Detlef</given-names></name><role>Senior Editor</role><aff><institution-wrap><institution-id institution-id-type="ror">https://ror.org/0243gzr89</institution-id><institution>Max Planck Institute for Biology Tübingen</institution></institution-wrap><country>Germany</country></aff></contrib></contrib-group><author-notes><fn fn-type="con" id="equal-contrib1"><label>†</label><p>These authors contributed equally to this work</p></fn><fn fn-type="present-address" id="pa1"><label>‡</label><p>Department of Applied Physics, Stanford University, Stanford, United States</p></fn><fn fn-type="present-address" id="pa2"><label>§</label><p>Department of Ecology and Genetics, Evolutionary Biology Center, Uppsala University, Uppsala, Sweden</p></fn><fn fn-type="present-address" id="pa3"><label>#</label><p>University Information Technology Services, University of Arizona, Tucson, United States</p></fn></author-notes><pub-date publication-format="electronic" date-type="publication"><day>06</day><month>09</month><year>2024</year></pub-date><volume>12</volume><elocation-id>RP87335</elocation-id><history><date date-type="sent-for-review" iso-8601-date="2023-03-16"><day>16</day><month>03</month><year>2023</year></date></history><pub-history><event><event-desc>This manuscript was published as a preprint.</event-desc><date date-type="preprint" iso-8601-date="2023-03-03"><day>03</day><month>03</month><year>2023</year></date><self-uri content-type="preprint" xlink:href="https://doi.org/10.1101/2023.03.02.530449"/></event><event><event-desc>This manuscript was published as a reviewed preprint.</event-desc><date date-type="reviewed-preprint" iso-8601-date="2023-07-26"><day>26</day><month>07</month><year>2023</year></date><self-uri content-type="reviewed-preprint" xlink:href="https://doi.org/10.7554/eLife.87335.1"/></event><event><event-desc>The reviewed preprint was revised.</event-desc><date date-type="reviewed-preprint" iso-8601-date="2024-07-16"><day>16</day><month>07</month><year>2024</year></date><self-uri content-type="reviewed-preprint" xlink:href="https://doi.org/10.7554/eLife.87335.2"/></event></pub-history><permissions><copyright-statement>© 2023, Weibel, Wheeler et al</copyright-statement><copyright-year>2023</copyright-year><copyright-holder>Weibel, Wheeler et al</copyright-holder><ali:free_to_read/><license xlink:href="http://creativecommons.org/licenses/by/4.0/"><ali:license_ref>http://creativecommons.org/licenses/by/4.0/</ali:license_ref><license-p>This article is distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="http://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution License</ext-link>, which permits unrestricted use and redistribution provided that the original author and source are credited.</license-p></license></permissions><self-uri content-type="pdf" xlink:href="elife-87335-v1.pdf"/><self-uri content-type="figures-pdf" xlink:href="elife-87335-figures-v1.pdf"/><abstract><p>The nearly neutral theory of molecular evolution posits variation among species in the effectiveness of selection. In an idealized model, the census population size determines both this minimum magnitude of the selection coefficient required for deleterious variants to be reliably purged, and the amount of neutral diversity. Empirically, an ‘effective population size’ is often estimated from the amount of putatively neutral genetic diversity and is assumed to also capture a species’ effectiveness of selection. A potentially more direct measure of the effectiveness of selection is the degree to which selection maintains preferred codons. However, past metrics that compare codon bias across species are confounded by among-species variation in %GC content and/or amino acid composition. Here, we propose a new Codon Adaptation Index of Species (CAIS), based on Kullback–Leibler divergence, that corrects for both confounders. We demonstrate the use of CAIS correlations, as well as the Effective Number of Codons, to show that the protein domains of more highly adapted vertebrate species evolve higher intrinsic structural disorder.</p></abstract><abstract abstract-type="plain-language-summary"><title>eLife digest</title><p>Evolution is the process through which populations change over time, starting with mutations in the genetic sequence of an organism. Many of these mutations harm the survival and reproduction of an organism, but only by a very small amount.</p><p>Some species, especially those with large populations, can purge these slightly harmful mutations more effectively than other species. This fact has been used by the ‘drift barrier theory’ to explain various profound differences amongst species, including differences in biological complexity. In this theory, the effectiveness of eliminating slightly harmful mutations is specified by an ‘effective' population size, which depends on factors beyond just the number of individuals in the population.</p><p>Effective population size is normally calculated from the amount of time a ‘neutral’ mutation (one with no effect at all) stays in the population before becoming lost or taking over. Estimating this time requires both representative data for genetic diversity and knowledge of the mutation rate. A major limitation is that these data are unavailable for most species. A second limitation is that a brief, temporary reduction in the number of individuals has an oversized impact on the metric, relative to its impact on the number of slighly harmful mutations accumulated.</p><p>Weibel, Wheeler et al. developed a new metric to more directly determine how effectively a species purges slightly harmful mutations. Their approach is based on the fact that the genetic code has ‘synonymous’ sequences. These sequences code for the same amino acid building block, with one of these sequences being only slightly preferred over others.</p><p>The metric by Weibel, Wheeler et al. quantifies the proportion of the genome from which less preferred synonymous sequences have been effectively purged. It judges a population to have a higher effective population size when the usage of synonymous sequences departs further from the usage predicted from mutational processes.</p><p>The researchers expected that natural selection would favour ‘ordered’ proteins with robust three-dimensional structures, i.e., that species with a higher effective population size would tend to have more ordered versions of a protein. Instead, they found the opposite: species with a higher effective population size tend to have more disordered versions of the same protein. This changes our view of how natural selection acts on proteins.</p><p>Why species are so different remains a fundamental question in biology. Weibel, Wheeler et al. provide a useful tool for future applications of drift barrier theory to a broad range of ways that species differ.</p></abstract><kwd-group kwd-group-type="author-keywords"><kwd>protein structural disorder</kwd><kwd>weak selection</kwd><kwd>drift barrier theory</kwd><kwd>codon usage bias</kwd></kwd-group><kwd-group kwd-group-type="research-organism"><title>Research organism</title><kwd>None</kwd></kwd-group><funding-group><award-group id="fund1"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100000002</institution-id><institution>National Institutes of Health</institution></institution-wrap></funding-source><award-id>GM104040</award-id><principal-award-recipient><name><surname>Weibel</surname><given-names>Catherine A</given-names></name><name><surname>James</surname><given-names>Jennifer E</given-names></name><name><surname>Willis</surname><given-names>Sara M</given-names></name><name><surname>Masel</surname><given-names>Joanna</given-names></name></principal-award-recipient></award-group><award-group id="fund2"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100000002</institution-id><institution>National Institutes of Health</institution></institution-wrap></funding-source><award-id>GM132008</award-id><principal-award-recipient><name><surname>Wheeler</surname><given-names>Andrew L</given-names></name></principal-award-recipient></award-group><award-group id="fund3"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100000925</institution-id><institution>John Templeton Foundation</institution></institution-wrap></funding-source><award-id>60814</award-id><principal-award-recipient><name><surname>Weibel</surname><given-names>Catherine A</given-names></name><name><surname>James</surname><given-names>Jennifer E</given-names></name><name><surname>Willis</surname><given-names>Sara M</given-names></name><name><surname>Masel</surname><given-names>Joanna</given-names></name></principal-award-recipient></award-group><award-group id="fund4"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100000997</institution-id><institution>Arnold and Mabel Beckman Foundation</institution></institution-wrap></funding-source><award-id>Scholars Program</award-id><principal-award-recipient><name><surname>Weibel</surname><given-names>Catherine A</given-names></name></principal-award-recipient></award-group><award-group id="fund5"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100000001</institution-id><institution>National Science Foundation</institution></institution-wrap></funding-source><award-id>WAESO/LSAMP Cooperative Agreement HRD-1101728</award-id><principal-award-recipient><name><surname>Weibel</surname><given-names>Catherine A</given-names></name></principal-award-recipient></award-group><award-group id="fund6"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100000104</institution-id><institution>National Aeronautics and Space Administration</institution></institution-wrap></funding-source><award-id>Arizona NASA Space Grant Consortium, Cooperative Agreement 80NSSC20M0041</award-id><principal-award-recipient><name><surname>Weibel</surname><given-names>Catherine A</given-names></name></principal-award-recipient></award-group><award-group id="fund7"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/501100008982</institution-id><institution>National Science Foundation</institution></institution-wrap></funding-source><award-id>Graduate Research Fellowship Program</award-id><principal-award-recipient><name><surname>McShea</surname><given-names>Hanon</given-names></name></principal-award-recipient></award-group><funding-statement>The funders had no role in study design, data collection, and interpretation, or the decision to submit the work for publication.</funding-statement></funding-group><custom-meta-group><custom-meta specific-use="meta-only"><meta-name>Author impact statement</meta-name><meta-value>Proteins evolve more intrinsic structural disorder under more effective selection, with selection assessed via a novel metric of codon adaptation.</meta-value></custom-meta><custom-meta specific-use="meta-only"><meta-name>publishing-route</meta-name><meta-value>prc</meta-value></custom-meta></custom-meta-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>Species differ from each other in many ways, including mating system, ploidy, spatial distribution, life history, size, lifespan, genome size, mutation rate, selective pressure, and population size. These differences make the process of purifying selection more efficient in some species than others. Our understanding of both the causes and consequences of these differences is limited in part by a reliable metric with which to measure them. In the long term, the probability that a gene is fixed for one allele rather than another allele is given by the ratio of fixation and counter-fixation probabilities (<xref ref-type="bibr" rid="bib8">Bulmer, 1991</xref>). In an idealized population of constant population size and no selection at linked sites, a mutation–selection–drift model describes how this ratio of fixation probabilities depends on the census population size <italic>N</italic> (<xref ref-type="bibr" rid="bib37">Kimura, 1962</xref>), and hence gives the fraction of sites expected to be found in preferred vs. non-preferred states (<xref ref-type="fig" rid="fig1">Figure 1</xref>).</p><fig id="fig1" position="float"><label>Figure 1.</label><caption><title>The effectiveness of selection, calculated as the long-term ratio of time spent in fixed deleterious: fixed beneficial allele states given symmetric mutation rates, is a function of the product <inline-formula><mml:math id="inf1"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>s</mml:mi><mml:mi>N</mml:mi></mml:mrow></mml:mstyle></mml:math></inline-formula>.</title><p>Assuming a diploid Wright–Fisher population with <italic>s</italic> &lt;&lt; 1, the probability of fixation of a new mutation <inline-formula><mml:math id="inf2"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>π</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>N</mml:mi><mml:mo>,</mml:mo><mml:mspace width="thinmathspace"/><mml:mi>s</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn><mml:mo>−</mml:mo><mml:msup><mml:mi>e</mml:mi><mml:mrow><mml:mo>−</mml:mo><mml:mspace width="thinmathspace"/><mml:mfrac><mml:mi>s</mml:mi><mml:mn>2</mml:mn></mml:mfrac></mml:mrow></mml:msup></mml:mrow><mml:mrow><mml:mn>1</mml:mn><mml:mo>−</mml:mo><mml:msup><mml:mi>e</mml:mi><mml:mrow><mml:mo>−</mml:mo><mml:mi>N</mml:mi><mml:mi>s</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:mfrac></mml:mrow></mml:mstyle></mml:math></inline-formula> , and the <italic>y</italic>-axis is calculated as <inline-formula><mml:math id="inf3"><mml:mrow><mml:mrow><mml:mi>π</mml:mi><mml:mfenced separators="|"><mml:mrow><mml:mi>N</mml:mi><mml:mo>,</mml:mo><mml:mo>-</mml:mo><mml:mi>s</mml:mi></mml:mrow></mml:mfenced></mml:mrow><mml:mo>/</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mi>π</mml:mi><mml:mfenced separators="|"><mml:mrow><mml:mi>N</mml:mi><mml:mo>,</mml:mo><mml:mo>-</mml:mo><mml:mi>s</mml:mi></mml:mrow></mml:mfenced><mml:mo>+</mml:mo><mml:mi>π</mml:mi><mml:mfenced separators="|"><mml:mrow><mml:mi>N</mml:mi><mml:mo>,</mml:mo><mml:mi>s</mml:mi></mml:mrow></mml:mfenced><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:math></inline-formula>. <italic>s</italic> is held constant at a value of 0.001 and <italic>N</italic> is varied. Results for other small magnitude values of <italic>s</italic> are superimposable. For small <inline-formula><mml:math id="inf4"><mml:mi>s</mml:mi><mml:mi>N</mml:mi></mml:math></inline-formula>, selection is ineffective at producing codon bias. For large <inline-formula><mml:math id="inf5"><mml:mi>s</mml:mi><mml:mi>N</mml:mi></mml:math></inline-formula>, selection is highly effective. For only a relatively narrow range of intermediate values of <inline-formula><mml:math id="inf6"><mml:mi>s</mml:mi><mml:mi>N</mml:mi></mml:math></inline-formula>, the degree of codon bias depends quantitatively on <inline-formula><mml:math id="inf7"><mml:mi>s</mml:mi><mml:mi>N</mml:mi></mml:math></inline-formula>.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-87335-fig1-v1.tif"/></fig><p>This reasoning has been extended to real populations by positing that species have an ‘effective’ population size, <italic>N<sub>e</sub></italic> (<xref ref-type="bibr" rid="bib53">Ohta, 1973</xref>). <italic>N<sub>e</sub></italic> is the census population size of an idealized population that reproduces a property of interest in the focal population. <italic>N<sub>e</sub></italic> is therefore not a single quantity per population, but instead depends on which property is of interest.</p><p>The amount of neutral polymorphism is the usual property used to empirically estimate <italic>N<sub>e</sub></italic> (<xref ref-type="bibr" rid="bib9">Charlesworth, 2009</xref>; <xref ref-type="bibr" rid="bib14">Doyle et al., 2015</xref>; <xref ref-type="bibr" rid="bib47">Lynch et al., 2016</xref>). However, the property of most relevance to nearly neutral theory is instead the inflection point <italic>s</italic> at which non-preferred alleles become common enough to matter (<xref ref-type="fig" rid="fig1">Figure 1</xref>), and hence the degree to which highly exquisite adaptation can be maintained in the face of ongoing mutation and genetic drift (<xref ref-type="bibr" rid="bib37">Kimura, 1962</xref>; <xref ref-type="bibr" rid="bib52">Ohta, 1972</xref>; <xref ref-type="bibr" rid="bib54">Ohta, 1992</xref>). While genetic diversity has been found to reflect some aspects of life history strategy (<xref ref-type="bibr" rid="bib60">Romiguier et al., 2014</xref>), there remain concerns about whether neutral genetic diversity and the limits to weak selection always remain closely coupled in non-equilibrium settings.</p><p>As a practical matter, <italic>N<sub>e</sub></italic> is usually calculated by dividing some measure of the amount of putatively neutral (often synonymous) polymorphism segregating in a population by that species’ mutation rate (<xref ref-type="bibr" rid="bib9">Charlesworth, 2009</xref>). As a result, <italic>N<sub>e</sub></italic> values are only available for species that have both polymorphism data and accurate mutation rate estimates, limiting their use. Worse, <italic>N<sub>e</sub></italic> is not a robust statistic. In the absence of a clear species definition, polymorphism is sometimes calculated across too broad a range of genomes, substantially inflating <italic>N<sub>e</sub></italic> (<xref ref-type="bibr" rid="bib11">Daubin and Moran, 2004</xref>); a poor sampling scheme can have the converse effect of deflating genetic diversity. Transient hypermutation (<xref ref-type="bibr" rid="bib56">Plotkin et al., 2006</xref>), which is common in microbes, causes further short-term inconsistencies in polymorphism levels. Perhaps most importantly, a recent bottleneck will deflate <italic>N<sub>e</sub></italic> based on the coalescence time, even if too brief to lead to significant erosion of fine-tuned adaptations. But drift barrier theory concerns the level with which adaptation is fine-tuned, and so a better metric would capture that directly, rather than indirectly rely on neutral diversity.</p><p>An alternative approach to measure the efficiency of selection exploits codon usage bias, which is influenced by weak selection for factors such as translational speed and accuracy (<xref ref-type="bibr" rid="bib29">Hershberg and Petrov, 2008</xref>; <xref ref-type="bibr" rid="bib57">Plotkin and Kudla, 2011</xref>; <xref ref-type="bibr" rid="bib33">Hunt et al., 2014</xref>). The degree of bias in synonymous codon usage that is driven by selective preference offers a more direct way to assess how effective selection is at the molecular level in a given species (<xref ref-type="bibr" rid="bib44">Li, 1987</xref>; <xref ref-type="bibr" rid="bib8">Bulmer, 1991</xref>; <xref ref-type="bibr" rid="bib2">Akashi, 1996</xref>; <xref ref-type="bibr" rid="bib67">Subramanian, 2008</xref>). Conveniently, it can be estimated from only a single genome, that is, without polymorphism or mutation rate data for that species.</p><p>One commonly used metric, the Codon Adaptation Index (CAI) (<xref ref-type="bibr" rid="bib63">Sharp and Li, 1987</xref>; <xref ref-type="bibr" rid="bib65">Sharp et al., 2010</xref>) takes the average of Relative Synonymous Codon Usage (RSCU) scores, which quantify how often a codon is used, relative to the codon that is most frequently used to encode that amino acid in that species. While this works well for comparing genes within the same species, it unfortunately means that the species-wide strength of codon bias appears in the normalizing denominator (see <xref ref-type="disp-formula" rid="equ4">Equation 4</xref> and <xref ref-type="fig" rid="fig3s1">Figure 3—figure supplement 1A</xref>). Paradoxically, this can make more exquisitely adapted species have lower rather than higher species-averaged CAI scores (<xref ref-type="fig" rid="fig3s1">Figure 3—figure supplement 1B</xref>; <xref ref-type="bibr" rid="bib58">Rocha, 2004</xref>; <xref ref-type="bibr" rid="bib6">Botzman and Margalit, 2011</xref>).</p><p>To compare species using CAI, it has been suggested that instead of taking a genome-wide average, one should consider a set of highly expressed reference genes (<xref ref-type="bibr" rid="bib64">Sharp et al., 2005</xref>; <xref ref-type="bibr" rid="bib72">Vicario et al., 2007</xref>; <xref ref-type="bibr" rid="bib67">Subramanian, 2008</xref>; <xref ref-type="bibr" rid="bib13">dos Reis and Wernisch, 2009</xref>). This approach assumes that the relative strength of selection on those reference genes (often a function of gene expression) remains approximately constant across the set of species considered (red distributions in <xref ref-type="fig" rid="fig2">Figure 2</xref>). Its use also requires careful attention to the length of reference genes (<xref ref-type="bibr" rid="bib70">Urrutia and Hurst, 2001</xref>; <xref ref-type="bibr" rid="bib12">Doherty and McInerney, 2013</xref>), and some approaches also require information about tRNA gene copy numbers and abundances (<xref ref-type="bibr" rid="bib13">dos Reis and Wernisch, 2009</xref>).</p><fig id="fig2" position="float"><label>Figure 2.</label><caption><title>More highly adapted species (bottom) have a higher proportion of their sites subject to effective selection on codon bias (blue area).</title><p>The Codon Adaptation Index (CAI) attempts to compare the intensity of selection (<xref ref-type="fig" rid="fig1">Figure 1</xref>, <italic>x</italic>-axis) in a subset of genes under strong selection (red areas). Given the narrow range of quantitative dependence of codon bias on <inline-formula><mml:math id="inf8"><mml:mi>s</mml:mi><mml:mi>N</mml:mi></mml:math></inline-formula> shown in <xref ref-type="fig" rid="fig1">Figure 1</xref>, our new metric is intended to capture differences in the proportion of the proteome subject to substantial selection (blue areas).</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-87335-fig2-v1.tif"/></fig><p>Since codon bias varies quantitatively within only a small range of <inline-formula><mml:math id="inf9"><mml:mi>s</mml:mi><mml:mi>N</mml:mi></mml:math></inline-formula> (<xref ref-type="fig" rid="fig1">Figure 1</xref>), a promising approach is to measure the proportion of sites at which codon adaptation is effective. We posit that more highly adapted species have a higher proportion of both genes and sites subject to effective selection on codon bias (<xref ref-type="fig" rid="fig2">Figure 2</xref>; <xref ref-type="bibr" rid="bib26">Galtier et al., 2018</xref>). Indeed, CAI might also rely in part on variation in the fraction of sites within the reference genes that is subject to effective selection as a function of species (<xref ref-type="fig" rid="fig2">Figure 2</xref>, red). Here we take this logic further, considering all sites in a proteome-wide approach. Averaging across the entire proteome provides robustness to shifts in the expression level of or strength of selection on particular genes. The proteome-wide average depends on the fraction of sites whose selection coefficients exceed the ‘drift barrier’ for that particular species (<xref ref-type="fig" rid="fig2">Figure 2</xref>, blue threshold).</p><p>In estimating the effects of selection, it is critical to control for other causes of codon bias. In particular, species differ in their mutational bias with respect to the proportion of the genome that consists of guanine-cytosine base pairs (GC), and in the frequency of GC-biased gene conversion (<xref ref-type="bibr" rid="bib70">Urrutia and Hurst, 2001</xref>; <xref ref-type="bibr" rid="bib17">Duret and Galtier, 2009</xref>; <xref ref-type="bibr" rid="bib12">Doherty and McInerney, 2013</xref>; <xref ref-type="bibr" rid="bib20">Figuet et al., 2014</xref>). Here, we control for %GC, capturing species differences both in mutation and in gene conversion, by calculating the Kullback–Leibler divergence of the observed codon frequencies away from the codon frequencies that we would expect to see given the genomic %GC content of the species. Kullback–Leibler divergence measures the distance of an observed probability distribution from an expected reference distribution, capturing a measure of surprise (<xref ref-type="bibr" rid="bib40">Kullback and Leibler, 1951</xref>). This method does not require us to specify preferred vs. non-preferred codons, and can thus also accommodate situations in which different genes have different codon preferences (<xref ref-type="bibr" rid="bib27">Gingold et al., 2014</xref>; <xref ref-type="bibr" rid="bib10">Cope et al., 2018</xref>).</p><p>An alternative metric, the Effective Number of Codons (ENC) originally quantified how far the codon usage of a sequence departs from equal usage of synonymous codons (<xref ref-type="bibr" rid="bib74">Wright, 1990</xref>), with lower ENC values indicating greater departure. This approach creates a complex relationship with GC content (<xref ref-type="bibr" rid="bib24">Fuglsang, 2008</xref>), and so ENC was later modified to correct for GC content (<xref ref-type="bibr" rid="bib50">Novembre, 2002</xref>). However, a remaining issue with this modified ENC is that differences among species in amino acid composition might act as a confounding factor, even after controlling for GC content. Specifically, species that make more use of an amino acid for which there is stronger selection among codons (which is sometimes the case <xref ref-type="bibr" rid="bib72">Vicario et al., 2007</xref>) would have higher codon bias, even if each amino acid considered on its own had identical codon bias irrespective of which species it is in. Confounding with amino acid frequencies has been shown to be a problem at the individual protein level (<xref ref-type="bibr" rid="bib10">Cope et al., 2018</xref>). Neither ENC (<xref ref-type="bibr" rid="bib23">Fuglsang, 2004</xref>; <xref ref-type="bibr" rid="bib24">Fuglsang, 2008</xref>) nor the CAI (<xref ref-type="bibr" rid="bib63">Sharp and Li, 1987</xref>) adequately control for differences in amino acid composition when applied across species. Despite early claims to the contrary (<xref ref-type="bibr" rid="bib74">Wright, 1990</xref>), this problem is not easy to fix for ENC (<xref ref-type="bibr" rid="bib23">Fuglsang, 2004</xref>; <xref ref-type="bibr" rid="bib24">Fuglsang, 2008</xref>).</p><p>Here, we extend the CAI, using the information-theory-based Kullback–Leibler divergence, so that it corrects for both GC and amino acid composition (see Methods) to create a new Codon Adaptation Index of Species (CAIS). The availability of a complete genome allows both metrics to be readily calculated without data on polymorphism or mutation rate, without selecting reference genes, and without concerns about demographic history. Our purpose is to find an accessible metric that can quantify the limits to weak selection important to nearly neutral theory; this differs from past evaluations focused on comparing different genes of the same species and recapitulating ‘ground truth’ simulations thereof (<xref ref-type="bibr" rid="bib68">Sun et al., 2013</xref>; <xref ref-type="bibr" rid="bib76">Zhang et al., 2012</xref>; <xref ref-type="bibr" rid="bib45">Liu et al., 2018</xref>). To demonstrate the usefulness of our method, we identify a novel correlation with intrinsic structural disorder (ISD), pointing to what else might be subject to weak selective preferences at the molecular level. While ENC can also identify subtle selection on ISD, CAIS can do so without the risk of confounding with amino acid frequencies.</p></sec><sec id="s2" sec-type="results"><title>Results</title><sec id="s2-1"><title>Both ENC and CAIS solve the GC confounding problem that plagues CAI</title><p>CAI is seriously confounded with GC content (<xref ref-type="fig" rid="fig3">Figure 3A</xref>). ENC is not confounded with GC content (<xref ref-type="fig" rid="fig3">Figure 3B</xref>), while CAIS has only a very weak correlation that is not significant after correction for multiple comparisons (<xref ref-type="fig" rid="fig3">Figure 3C</xref>).</p><fig-group><fig id="fig3" position="float"><label>Figure 3.</label><caption><title>Codon Adaptation Index (CAI) is seriously confounded with GC content (<bold>A</bold>), while Effective Number of Codons (ENC) and Codon Adaptation Index of Species (CAIS) are not (<bold>B</bold> and <bold>C</bold>).</title><p>We control for phylogenetic confounding via Phylogenetic Independent Contrasts (PIC) (<xref ref-type="bibr" rid="bib19">Felsenstein, 1985</xref>); this yields an unbiased <italic>R</italic><sup>2</sup> estimate (<xref ref-type="bibr" rid="bib59">Rohlf, 2006</xref>). Each datapoint is one of 118 vertebrate species with ‘Complete’ intergenic genomic sequence (allowing for %GC correction) and TimeTree divergence dates (allowing for PIC correction). Red line shows unweighted lm(<italic>y</italic> ~ <italic>x</italic>) with gray region as 95% confidence interval. <xref ref-type="fig" rid="fig3s1">Figure 3—figure supplement 1</xref> shows in more detail why CAI is not appropriate for species-wide effectiveness of selection measurements. Plots without PIC correction are shown in <xref ref-type="fig" rid="fig3s2">Figure 3—figure supplement 2</xref>. The impact of amino acid frequency correction on CAIS is shown in <xref ref-type="fig" rid="fig3s3">Figure 3—figure supplement 3</xref>.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-87335-fig3-v1.tif"/></fig><fig id="fig3s1" position="float" specific-use="child-fig"><label>Figure 3—figure supplement 1.</label><caption><title>Codon Adaptation Index (CAI) is not appropriate for species-wide effectiveness of selection measurements.</title><p>Each CAI value shown is averaged over an entire species’ proteome. (<bold>A</bold>) The value of CAI is driven by its normalizing denominator term, CAI<sub>max</sub>. (<bold>B</bold>) As a result, CAI is inversely proportional to Codon Adaptation Index of Species (CAIS). Each datapoint is one of 118 vertebrate species with ‘Complete’ intergenic genomic sequence available (allowing for %GC correction) and TimeTree divergence dates (allowing for Phylogenetic Independent Contrasts [PIC] correction). p-values shown are for Pearson’s correlation.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-87335-fig3-figsupp1-v1.tif"/></fig><fig id="fig3s2" position="float" specific-use="child-fig"><label>Figure 3—figure supplement 2.</label><caption><title>The same relationships are shown as in <xref ref-type="fig" rid="fig3">Figure 3</xref>, but without correction for phylogenetic confounding, suggesting GC confounding for the Effective Number of Codons (ENC) but not the Codon Adaptation Index of Species (CAIS).</title><p>Codon Adaptation Index (CAI) (<bold>A</bold>) and ENC (<bold>B</bold>) both correlate with genomic GC, but CAIS (<bold>C</bold>) does not. Red line shows lm(<italic>y</italic> ~ <italic>x</italic>), with gray region as 95% confidence interval. We use Phylogenetic Independent Contrasts (PIC) corrected results rather than these results because PIC correction removes non-independent errors to produce an unbiased <italic>R</italic><sup>2</sup> estimate (<xref ref-type="bibr" rid="bib59">Rohlf, 2006</xref>).</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-87335-fig3-figsupp2-v1.tif"/></fig><fig id="fig3s3" position="float" specific-use="child-fig"><label>Figure 3—figure supplement 3.</label><caption><title>Vertebrate Codon Adaptation Index of Species (CAIS) values are not greatly affected by computation for a standardized amino acid composition vs. computation for the amino acid frequencies in the species in question.</title></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-87335-fig3-figsupp3-v1.tif"/></fig></fig-group></sec><sec id="s2-2"><title>Proteins in better adapted species evolve more structural disorder</title><p>As an example of how correlations with codon adaptation metrics can be used to identify weak selective preferences, we investigate protein ISD. Disordered proteins are more likely to be harmful when overexpressed (<xref ref-type="bibr" rid="bib71">Vavouri et al., 2009</xref>), and ISD is more abundant in eukaryotic than prokaryotic proteins (<xref ref-type="bibr" rid="bib62">Schad et al., 2011</xref>; <xref ref-type="bibr" rid="bib75">Xue et al., 2012</xref>; <xref ref-type="bibr" rid="bib3">Basile et al., 2019</xref>), suggesting that low ISD might be favored by more effective selection.</p><p>However, compositional differences among proteomes might not be driven by differences in how a given protein sequence evolves as a function of the effectiveness of selection. Instead, they might be driven by the recent birth of ISD-rich proteins in animals (<xref ref-type="bibr" rid="bib34">James et al., 2021</xref>), and/or by differences among sequences in their subsequent tendency to proliferate into many different genes (<xref ref-type="bibr" rid="bib35">James et al., 2023</xref>). To focus only on the effects of descent with modification, we use a linear mixed model, with each species having a fixed effect on ISD, while controlling for Pfam domain identity as a random effect. We note that once GC is controlled for, codon adaptation can be assessed similarly in intrinsically disordered vs. ordered proteins (<xref ref-type="bibr" rid="bib28">Gossmann et al., 2012</xref>). Controlling for Pfam identity is supported, with standard deviation in ISD of 0.178 among Pfams compared to residual standard deviation of 0.058, and a p-value on the significance of the Pfam random effect term of 3 × 10<sup>−13</sup>. Controlling in this way for Pfam identity, we then ask whether the fixed species effects on ISD are correlated with CAIS and with ENC.</p><p>Surprisingly, more exquisitely adapted species have more disordered protein domains (<xref ref-type="fig" rid="fig4">Figure 4</xref>). Results using ENC and CAIS are similar, with ENC having higher power; the correlation coefficient is 0.36 for CAIS compared to 0.50 for ENC, and the p-value for ENC is 3 orders of magnitude lower. We note, however, that amino acid frequencies strongly influence ISD (<xref ref-type="bibr" rid="bib69">Theillet et al., 2013</xref>). The CAIS correlation is more reliable than the ENC correlation because by construction, CAIS controls for differences in amino acid frequencies among species.</p><fig-group><fig id="fig4" position="float"><label>Figure 4.</label><caption><title>Protein domains have higher intrinsic structural disorder (ISD) when found in more exquisitely adapted species, according to (<bold>A</bold>) the Codon Adaptation Index of Species (CAIS) and (<bold>B</bold>) the Effective Number of Codons (ENC).</title><p>We plot -ENC rather than ENC to more easily compare results with those from CAIS. (<bold>C</bold>) Correcting for local rather than genome-wide %GC removes the relationship. Each datapoint is one of 118 vertebrate species with ‘complete’ intergenic genomic sequence available (allowing for %GC correction), and TimeTree divergence dates (allowing for Phylogenetic Independent Contrasts [PIC] correction). ‘Effects’ on ISD shown on the <italic>y</italic>-axis are fixed effects of species identity in our linear mixed model, after PIC correction. Red line shows unweighted lm(<italic>y</italic> ~ <italic>x</italic>) with gray region as 95% confidence interval. Panels without PIC correction are presented in <xref ref-type="fig" rid="fig4s1">Figure 4—figure supplement 1</xref>.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-87335-fig4-v1.tif"/></fig><fig id="fig4s1" position="float" specific-use="child-fig"><label>Figure 4—figure supplement 1.</label><caption><title>The same relationships are shown as in <xref ref-type="fig" rid="fig4">Figure 4</xref>, here without correction for phylogenetic confounding.</title><p>As in <xref ref-type="fig" rid="fig4">Figure 4</xref>, intrinsic structural disorder (ISD) of protein domains is higher in more highly adapted species, as measured by Codon Adaptation Index of Species (CAIS) (<bold>A</bold>) and Effective Number of Codons (ENC) (<bold>B</bold>), but not by CAIS calculated with local GC% rather than genome-wide GC% (<bold>C</bold>). ISD is calculated as in <xref ref-type="fig" rid="fig4">Figure 4</xref>. Red line shows lm(<italic>y</italic> ~ <italic>x</italic>), with gray region as 95% confidence interval.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-87335-fig4-figsupp1-v1.tif"/></fig></fig-group><p>Different parts of the genome have different GC contents (<xref ref-type="bibr" rid="bib4">Bernardi, 2000</xref>; <xref ref-type="bibr" rid="bib18">Eyre-Walker and Hurst, 2001</xref>; <xref ref-type="bibr" rid="bib42">Lander et al., 2001</xref>), primarily because the extent to which GC-biased gene conversion increases GC content depends on the local rate of recombination (<xref ref-type="bibr" rid="bib25">Galtier et al., 2001</xref>; <xref ref-type="bibr" rid="bib49">Meunier and Duret, 2004</xref>; <xref ref-type="bibr" rid="bib16">Duret et al., 2006</xref>; <xref ref-type="bibr" rid="bib17">Duret and Galtier, 2009</xref>). We therefore also calculated a version of CAIS whose codon frequency expectations are based on local intergenic GC content. This performed worse (<xref ref-type="fig" rid="fig4">Figure 4C</xref>) than our simple use of genome-wide GC content (<xref ref-type="fig" rid="fig4">Figure 4A</xref>) with respect to the strength of correlation between CAIS and ISD. If GC-biased gene conversion is a more powerful force than weak selective preferences among codons, then local GC content will evolve more rapidly than codon usage (<xref ref-type="bibr" rid="bib38">Kondrashov et al., 2010</xref>). In this case, genome-wide GC may serve as an appropriately time-averaged proxy. It is also possible that the local non-coding sequences we used were too short (at 3000 bp or more), creating excessive noise that obscured the signal.</p><p>Many vertebrates have higher recombination rates and hence GC-biased gene conversion near genes; in this case genome-wide GC content would misestimate the codon usage expected from the combination of mutation bias and GC-biased gene conversion in the vicinity of genes. If GC-biased gene conversion drove CAIS, we expect high <inline-formula><mml:math id="inf10"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow><mml:mover><mml:mrow><mml:mrow><mml:mi mathvariant="normal">l</mml:mi><mml:mi mathvariant="normal">o</mml:mi><mml:mi mathvariant="normal">c</mml:mi><mml:mi mathvariant="normal">a</mml:mi><mml:mi mathvariant="normal">l</mml:mi></mml:mrow><mml:mspace width="thinmathspace"/><mml:mrow><mml:mi mathvariant="normal">G</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">C</mml:mi></mml:mrow></mml:mrow><mml:mo accent="false">¯</mml:mo></mml:mover><mml:mo>−</mml:mo><mml:mrow><mml:mtext> </mml:mtext></mml:mrow><mml:mrow><mml:mi mathvariant="normal">g</mml:mi><mml:mi mathvariant="normal">l</mml:mi><mml:mi mathvariant="normal">o</mml:mi><mml:mi mathvariant="normal">b</mml:mi><mml:mi mathvariant="normal">a</mml:mi><mml:mi mathvariant="normal">l</mml:mi></mml:mrow><mml:mspace width="thinmathspace"/><mml:mrow><mml:mi mathvariant="normal">G</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">C</mml:mi></mml:mrow><mml:mrow><mml:mo stretchy="false">|</mml:mo></mml:mrow></mml:mrow></mml:mstyle></mml:math></inline-formula> to predict high CAIS. We do not see this relationship (<xref ref-type="fig" rid="fig5">Figure 5</xref>), suggesting that gene conversion strength is not a confounding factor impacting CAIS.</p><fig id="fig5" position="float"><label>Figure 5.</label><caption><title>Codon Adaptation Index of Species (CAIS) is not correlated with the degree to which local genomic regions differ in their GC content from global GC content.</title><p>If CAIS were driven by GC-biased gene conversion, genomes with more heterogeneous %GC distributions should have higher CAIS scores.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-87335-fig5-v1.tif"/></fig><p>Younger animal-specific protein domains have higher ISD (<xref ref-type="bibr" rid="bib34">James et al., 2021</xref>). It is possible that selection in favor of high ISD is strongest in young domains, which might use more primitive methods to avoid aggregation (<xref ref-type="bibr" rid="bib22">Foy et al., 2019</xref>; <xref ref-type="bibr" rid="bib5">Bertram and Masel, 2020</xref>). To test this, we analyze two subsets of our data: those that emerged prior to the last eukaryotic common ancestor (LECA), here referred to as ‘old’ protein domains, and ‘young’ protein domains that emerged after the divergence of animals and fungi from plants. Young and old domains show equally strong trends of increasing disorder with species’ adaptedness (<xref ref-type="fig" rid="fig6">Figure 6</xref>).</p><fig-group><fig id="fig6" position="float"><label>Figure 6.</label><caption><title>More exquisitely adapted species have higher intrinsic structural disorder (ISD) in both young (<bold>A</bold> and <bold>B</bold>) and old (<bold>C</bold> and <bold>D</bold>) protein domains, according to both the Codon Adaptation Index of Species (CAIS) (<bold>A, C</bold>), and the Effective Number of Codons (ENC) (<bold>B, D</bold>).</title><p>Age assignments are taken from <xref ref-type="bibr" rid="bib34">James et al., 2021</xref>, with vertebrate protein domains that emerged prior to last eukaryotic common ancestor (LECA) classified as ‘old’, and vertebrate protein domains that emerged after the divergence of animals and fungi from plants as ‘young’. ‘Effects’ on ISD shown on the <italic>y</italic>-axis are fixed effects of species identity in our linear mixed model. The same <italic>n</italic> = 118 datapoints are shown as in <xref ref-type="fig" rid="fig3">Figures 3</xref> and <xref ref-type="fig" rid="fig4">4</xref>. Red line shows lm(<italic>y</italic> ~ <italic>x</italic>), with gray region as 95% confidence interval. Panels without Phylogenetic Independent Contrasts (PIC) correction are shown in <xref ref-type="fig" rid="fig6s1">Figure 6—figure supplement 1</xref>.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-87335-fig6-v1.tif"/></fig><fig id="fig6s1" position="float" specific-use="child-fig"><label>Figure 6—figure supplement 1.</label><caption><title>Without correction for phylogenetic confounding, more highly adapted species have higher intrinsic structural disorder (ISD) in both young (<bold>A</bold> and <bold>B</bold>) and old (<bold>C</bold> and <bold>D</bold>) protein domains, according to both the Codon Adaptation Index of Species (CAIS) (<bold>A, C</bold>), and the Effective Number of Codons (ENC) (<bold>B, D</bold>).</title><p>Age assignments and ISD effects are calculated as in <xref ref-type="fig" rid="fig6">Figure 6</xref>. Same <italic>n</italic> = 118 datapoints are shown as in <xref ref-type="fig" rid="fig3">Figures 3</xref>—<xref ref-type="fig" rid="fig5">5</xref>. Red line shows lm(<italic>y</italic> ~ <italic>x</italic>), with gray region as 95% confidence interval.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-87335-fig6-figsupp1-v1.tif"/></fig></fig-group></sec></sec><sec id="s3" sec-type="discussion"><title>Discussion</title><p>When different properties are each causally affected by a species’ exquisiteness of adaptation, this will create a correlation between the properties. We use codon adaptation as a reference property, such that correlations with codon adaptation indicate selection. To detect ISD as a novel property under selection, we used a linear mixed model approach that controls for Pfam identity as a random effect. This approach shows that the same Pfam domain tends to be more disordered when found in a well-adapted species (i.e. a species with a higher CAIS or ENC). This is true for both ancient and recently emerged protein domains.</p><p>It is important that no additional variable such as GC content or amino acid frequencies creates a spurious correlation by affecting both CAIS and our property of interest. For this reason, we define CAIS as the observed Kullback–Leibler divergence (<xref ref-type="bibr" rid="bib40">Kullback and Leibler, 1951</xref>) from the codon usage expected given the GC content. The GC content pertinent to this expectation depends primarily on mutation bias and GC-biased gene conversion (<xref ref-type="bibr" rid="bib61">Romiguier and Roux, 2017</xref>), but potentially also on selection on individual nucleotide substitutions that is hypothesized to favor higher %GC (<xref ref-type="bibr" rid="bib46">Long et al., 2018</xref>). By controlling for %GC, we exclude all these forces from influencing CAIS or ENC. We thus capture the extent of adaptation in codon bias, including translational speed, accuracy, and any intrinsic preference for GC over AT that is specific to coding regions. These remaining codon-adaptive factors do not create a statistically convincing correlation between CAIS and GC (<xref ref-type="fig" rid="fig3">Figure 3C</xref>), nor between ENC and GC (<xref ref-type="fig" rid="fig3">Figure 3B</xref>), although CAI is strongly correlated with GC (<xref ref-type="fig" rid="fig3">Figure 3A</xref>). Notably, our new CAIS metric of codon adaptation controls for amino acid frequencies, rather than, like ENC, only GC content.</p><p>A direct effect of ISD on fitness agrees with studies of random Open Reading Frames (ORFs) in <italic>Escherichia coli</italic>, where fitness was driven more by amino acid composition than %GC content, after controlling for the intrinsic correlation between the two (<xref ref-type="bibr" rid="bib39">Kosinski et al., 2022</xref>). However, we have not ruled out a role for selection for higher %GC in ways that are general rather than restricted to coding regions, whether in shaping mutational biases (<xref ref-type="bibr" rid="bib66">Smith and Eyre-Walker, 2001</xref>; <xref ref-type="bibr" rid="bib30">Hershberg and Petrov, 2009</xref>; <xref ref-type="bibr" rid="bib31">Hildebrand et al., 2010</xref>; <xref ref-type="bibr" rid="bib51">Novoa et al., 2019</xref>; <xref ref-type="bibr" rid="bib21">Forcelloni and Giansanti, 2020</xref>) or the extent of gene conversion, or even at the single-nucleotide level in a manner shared between coding regions and intergenic regions (<xref ref-type="bibr" rid="bib46">Long et al., 2018</xref>).</p><p>A more complex metric could control for more than just GC content and amino acid frequencies. First vs. second vs. third codon positions have different nucleotide usage on average, but while correcting for this might be useful for comparing genes (<xref ref-type="bibr" rid="bib76">Zhang et al., 2012</xref>), correcting for it while comparing species might remove the effect of interest. Similarly, while it might be useful to control for dinucleotide and trinucleotide frequencies (<xref ref-type="bibr" rid="bib7">Brbić et al., 2015</xref>), to avoid circularity these would need to be taken from intergenic sequences, with care needed to avoid influence from unannotated protein-coding genes or even pseudogenes.</p><p>Note that if a species were to experience a sudden reduction in census population size, for example due to habitat loss, leading to less effective selection, it would take some multiple of the neutral coalescent time for CAIS to fully adjust. CAIS thus represents a relatively long-term historical pattern of adaptation. The timescales setting neutral polymorphism-based <inline-formula><mml:math id="inf11"><mml:msub><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>e</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> estimates are likely shorter, based on a single round of coalescence. It is possible that the reason that we obtained correlations when we controlled for genome-wide GC content, but not when we controlled for local GC content, is also that codon adaptation adjusts slowly relative to the timescale of fluctuations in local GC content.</p><p>Here, we developed a new metric of species adaptedness at the codon level, capable of quantifying degrees of codon adaptation even among vertebrates. We chose vertebrates partly due to the abundance of suitable data, and partly as a stringent test case, given past studies finding limited evidence for codon adaptation (<xref ref-type="bibr" rid="bib36">Kessler and Dean, 2014</xref>). It remains to be seen how CAIS behaves among species with stronger codon adaptation. We restricted our analysis to only the best annotated genomes, in part to ensure the quality of intergenic %GC estimates, and in part limited by the feasibility of running linear mixed models with 6 million datapoints. The phylogenetic tree is well resolved for vertebrate species, with an overrepresentation of mammalian species. Despite the focus on vertebrates, we were able to discover new results regarding selection on ISD.</p><p>Our finding that more effective selection prefers higher ISD was unexpected, given that lower-<italic>N<sub>e</sub></italic> eukaryotes have more disordered proteins than higher-<italic>N<sub>e</sub></italic> prokaryotes (<xref ref-type="bibr" rid="bib1">Ahrens et al., 2017</xref>; <xref ref-type="bibr" rid="bib3">Basile et al., 2019</xref>). However, this can be reconciled in a model in which highly disordered sequences are less likely to be found in high-<italic>N<sub>e</sub></italic> species, but the sequences that are present tend to have slightly higher disorder than their low-<italic>N<sub>e</sub></italic> homologs. High ISD might help mitigate the trade-off between affinity and specificity in protein–protein interactions (<xref ref-type="bibr" rid="bib15">Dunker et al., 1998</xref>; <xref ref-type="bibr" rid="bib32">Huang and Liu, 2013</xref>; <xref ref-type="bibr" rid="bib43">Lazar et al., 2022</xref>); non-specific interactions might be short-lived due to the high entropy associated with disorder, which specific interactions are robust to.</p><p>Codon adaptation metrics more directly quantify how species vary in their exquisiteness of adaptation, than do estimates of effective population size that are based on neutral polymorphism. Both CAIS and ENC can also be estimated for far more species because they do not require polymorphism or mutation rate data, nor tRNA gene copy numbers and abundances, but only a single complete genome. CAIS has the additional advantage of not being confounded with amino acid frequencies. This makes CAIS a useful tool for applying nearly neutral theory to protein evolution, as shown by our worked example of ISD.</p></sec><sec id="s4" sec-type="methods"><title>Methods</title><table-wrap id="keyresource" position="anchor"><label>Key resources table</label><table frame="hsides" rules="groups"><thead><tr><th align="left" valign="bottom">Reagent type (species) or resource</th><th align="left" valign="bottom">Designation</th><th align="left" valign="bottom">Source or reference</th><th align="left" valign="bottom">Identifiers</th><th align="left" valign="bottom">Additional information</th></tr></thead><tbody><tr><td align="left" valign="bottom">Software, algorithm</td><td align="left" valign="bottom">IUPRED2</td><td align="left" valign="bottom">DOI: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1093/nar/gky384">https://doi.org/10.1093/nar/gky384</ext-link></td><td align="left" valign="bottom">RRID:<ext-link ext-link-type="uri" xlink:href="https://identifiers.org/RRID/RRID:SCR_014632">SCR_014632</ext-link></td><td align="left" valign="bottom"/></tr><tr><td align="left" valign="bottom">Software, algorithm</td><td align="left" valign="bottom">Codon Adaptation Index of Species</td><td align="left" valign="bottom">This paper</td><td align="left" valign="bottom"/><td align="left" valign="bottom">See Materials and methods</td></tr><tr><td align="left" valign="bottom">Software, algorithm</td><td align="left" valign="bottom">Codon Adaptation Index</td><td align="left" valign="bottom">DOI: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1093/nar/15.3.1281">https://doi.org/10.1093/nar/15.3.1281</ext-link></td><td align="left" valign="bottom"/><td align="left" valign="bottom"/></tr><tr><td align="left" valign="bottom">Software, algorithm</td><td align="left" valign="bottom">ape</td><td align="left" valign="bottom">DOI: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1093/bioinformatics/bty633">https://doi.org/10.1093/bioinformatics/bty633</ext-link></td><td align="left" valign="bottom">RRID:<ext-link ext-link-type="uri" xlink:href="https://identifiers.org/RRID/RRID:SCR_017343">SCR_017343</ext-link></td><td align="left" valign="bottom">R package</td></tr><tr><td align="left" valign="bottom">Software, algorithm</td><td align="left" valign="bottom">Effective Number of Codons</td><td align="left" valign="bottom">DOI: <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1093/oxfordjournals.molbev.a004201">https://doi.org/10.1093/oxfordjournals.molbev.a004201</ext-link></td><td align="left" valign="bottom"/><td align="left" valign="bottom"/></tr></tbody></table></table-wrap><sec id="s4-1"><title>Species</title><p>Pfam sequences and IUPRED2 estimates of ISD predictions were taken from <xref ref-type="bibr" rid="bib34">James et al., 2021</xref>, who studied species marked as ‘Complete’ in the GOLD database, with divergence dates available in TimeTree (<xref ref-type="bibr" rid="bib41">Kumar et al., 2017</xref>). <xref ref-type="bibr" rid="bib34">James et al., 2021</xref> applied a variety of quality controls to exclude contaminants from the set of Pfams and assign accurate dates of Pfam emergence. Pfams that emerged prior to LECA are classified here as ‘old’, and Pfams that emerged after the divergence of animals and fungi from plants are classified as ‘young’, following annotation by <xref ref-type="bibr" rid="bib34">James et al., 2021</xref>. Species list and other information can be found at <ext-link ext-link-type="uri" xlink:href="https://github.com/MaselLab/Codon-Adaptation-Index-of-Species">https://github.com/MaselLab/Codon-Adaptation-Index-of-Species</ext-link> (copy archived at <xref ref-type="bibr" rid="bib48">MaselLab, 2024</xref>).</p></sec><sec id="s4-2"><title>Codon Adaptation Index</title><p><xref ref-type="bibr" rid="bib63">Sharp and Li, 1987</xref> quantified codon bias through the CAI, a normalized geometric mean of synonymous codon usage bias across sites, excluding stop and start codons. We modify this to calculate CAI including stop and start codons, because of documented preferences among stop codons in mammals (<xref ref-type="bibr" rid="bib73">Wangen and Green, 2020</xref>). While usually used to compare genes within a species, among-species comparisons can be made using a reference set of genes that are highly expressed (<xref ref-type="bibr" rid="bib63">Sharp and Li, 1987</xref>). Each codon <italic>i</italic> is assigned an RSCU value:<disp-formula id="equ1"><label>(1)</label><mml:math id="m1"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mi>R</mml:mi><mml:mi>S</mml:mi><mml:mi>C</mml:mi><mml:msub><mml:mi>U</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mfrac><mml:msub><mml:mi>N</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mfrac><mml:mn>1</mml:mn><mml:msub><mml:mi>n</mml:mi><mml:mrow><mml:mi>a</mml:mi></mml:mrow></mml:msub></mml:mfrac><mml:munderover><mml:mo>∑</mml:mo><mml:mrow><mml:mi>j</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:msub><mml:mi>n</mml:mi><mml:mrow><mml:mi>a</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:munderover><mml:msub><mml:mi>N</mml:mi><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfrac><mml:mo>,</mml:mo></mml:mstyle></mml:mrow></mml:math></disp-formula></p><p>where <inline-formula><mml:math id="inf12"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:msub><mml:mi>N</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mstyle></mml:math></inline-formula> denotes the number of times that codon <inline-formula><mml:math id="inf13"><mml:mi>i</mml:mi></mml:math></inline-formula> is used, and the denominator sums over all <inline-formula><mml:math id="inf14"><mml:msub><mml:mrow><mml:mi>n</mml:mi></mml:mrow><mml:mrow><mml:mi>a</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> codons that code for that specific amino acid. RSCU values are normalized to produce a relative adaptiveness values <inline-formula><mml:math id="inf15"><mml:msub><mml:mrow><mml:mi>w</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> for each codon, relative to the best adapted codon for that amino acid:<disp-formula id="equ2"><label>(2)</label><mml:math id="m2"><mml:mrow><mml:msub><mml:mi>w</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>≡</mml:mo><mml:mfrac><mml:mrow><mml:mi>R</mml:mi><mml:mi>S</mml:mi><mml:mi>C</mml:mi><mml:msub><mml:mi>U</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:mi>R</mml:mi><mml:mi>S</mml:mi><mml:mi>C</mml:mi><mml:msub><mml:mi>U</mml:mi><mml:mrow><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>x</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfrac><mml:mo>.</mml:mo></mml:mrow></mml:math></disp-formula></p><p>Let <inline-formula><mml:math id="inf16"><mml:mi>L</mml:mi></mml:math></inline-formula> be the number of codons across all protein-coding sequences considered. Then<disp-formula id="equ3"><label>(3)</label><mml:math id="m3"><mml:mrow><mml:mi>C</mml:mi><mml:mi>A</mml:mi><mml:mi>I</mml:mi><mml:mo>=</mml:mo><mml:msup><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:msubsup><mml:mi mathvariant="normal">Π</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>L</mml:mi></mml:mrow></mml:msubsup><mml:msub><mml:mi>w</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>]</mml:mo></mml:mrow><mml:mrow><mml:mfrac><mml:mn>1</mml:mn><mml:mi>L</mml:mi></mml:mfrac></mml:mrow></mml:msup><mml:mo>.</mml:mo></mml:mrow></mml:math></disp-formula></p><p>To understand the effects of normalization, it is useful to rewrite this as:<disp-formula id="equ4"><label>(4)</label><mml:math id="m4"><mml:mrow><mml:mi>C</mml:mi><mml:mi>A</mml:mi><mml:mi>I</mml:mi><mml:mo>=</mml:mo><mml:msup><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:msubsup><mml:mi mathvariant="normal">Π</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>L</mml:mi></mml:mrow></mml:msubsup><mml:mfrac><mml:mrow><mml:mi>R</mml:mi><mml:mi>S</mml:mi><mml:mi>C</mml:mi><mml:msub><mml:mi>U</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:mi>R</mml:mi><mml:mi>S</mml:mi><mml:mi>C</mml:mi><mml:msub><mml:mi>U</mml:mi><mml:mrow><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>x</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfrac></mml:mrow><mml:mo>]</mml:mo></mml:mrow><mml:mrow><mml:mfrac><mml:mn>1</mml:mn><mml:mi>L</mml:mi></mml:mfrac></mml:mrow></mml:msup><mml:mspace width="thinmathspace"/><mml:mo>=</mml:mo><mml:mspace width="thinmathspace"/><mml:mfrac><mml:mrow><mml:mi>C</mml:mi><mml:mi>A</mml:mi><mml:msub><mml:mi>I</mml:mi><mml:mrow><mml:mi>r</mml:mi><mml:mi>a</mml:mi><mml:mi>w</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:mi>C</mml:mi><mml:mi>A</mml:mi><mml:msub><mml:mi>I</mml:mi><mml:mrow><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>x</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfrac><mml:mo>,</mml:mo></mml:mrow></mml:math></disp-formula></p><p>where <inline-formula><mml:math id="inf17"><mml:msub><mml:mrow><mml:mi>C</mml:mi><mml:mi>A</mml:mi><mml:mi>I</mml:mi></mml:mrow><mml:mrow><mml:mi>r</mml:mi><mml:mi>a</mml:mi><mml:mi>w</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> is the geometric mean of the ‘unnormalized’ or observed synonymous codon usages, and <inline-formula><mml:math id="inf18"><mml:msub><mml:mrow><mml:mi>C</mml:mi><mml:mi>A</mml:mi><mml:mi>I</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mi>a</mml:mi><mml:mi>x</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> is the maximum possible CAI given the observed codon frequencies.</p></sec><sec id="s4-3"><title>GC content</title><p>We calculated total %GC content (intergenic and genic) during a scan of all six reading frames across genic and intergenic sequences available from NCBI with access dates between May and July 2019 (described in <xref ref-type="bibr" rid="bib35">James et al., 2023</xref>). Of the 170 vertebrates meeting the quality criteria of <xref ref-type="bibr" rid="bib34">James et al., 2021</xref>, 118 had annotated intergenic sequences within NCBI, so we restricted the dataset further to keep only the 118 species for which total GC content was available.</p></sec><sec id="s4-4"><title>Codon Adaptation Index of Species</title><sec id="s4-4-1"><title>Controlling for GC bias in synonymous codon usage</title><p>Consider a sequence region <inline-formula><mml:math id="inf19"><mml:mi>r</mml:mi></mml:math></inline-formula> within species <inline-formula><mml:math id="inf20"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>s</mml:mi></mml:mrow></mml:mstyle></mml:math></inline-formula> where each nucleotide has an expected probability of being G or C = <inline-formula><mml:math id="inf21"><mml:msub><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mrow><mml:mi>r</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>. For our main analysis, we consider just one region <inline-formula><mml:math id="inf22"><mml:mi>r</mml:mi></mml:math></inline-formula> encompassing the entire genome of a species <inline-formula><mml:math id="inf23"><mml:mi>s</mml:mi></mml:math></inline-formula>. In a secondary analysis, we break the genome up and use local values of <inline-formula><mml:math id="inf24"><mml:msub><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mrow><mml:mi>r</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> in the non-coding regions within and surrounding a gene or set of overlapping genes. To annotate the boundaries of these local regions, we first selected 1500 base pairs flanking each side of every coding sequence identified by NCBI annotations. Coding sequence annotations are broken up according to exon by NCBI. When coding sequences of the same gene did not fall within 3000 base pairs of each other, they were treated as different regions. When two coding sequences, whether from the same gene or from different genes, had overlapping 1500 bp catchment areas, we merged them together. <inline-formula><mml:math id="inf25"><mml:msub><mml:mrow><mml:mi>g</mml:mi></mml:mrow><mml:mrow><mml:mi>r</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> was then calculated based on the non-coding sites within each region, including both genic regions such as promoters and non-genic regions such as introns and intergenic sequences.</p><p>With no bias between C vs. G, nor between A vs. T, nor patterns beyond the overall composition taken one nucleotide at a time, the expected probability of seeing codon <inline-formula><mml:math id="inf26"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:mstyle></mml:math></inline-formula> in a triplet within <inline-formula><mml:math id="inf27"><mml:mi>r</mml:mi></mml:math></inline-formula> is<disp-formula id="equ5"><label>(5)</label><mml:math id="m5"><mml:mrow><mml:msub><mml:mi>p</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>r</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mtext> </mml:mtext><mml:msup><mml:mfrac><mml:msub><mml:mi>g</mml:mi><mml:mrow><mml:mi>r</mml:mi></mml:mrow></mml:msub><mml:mn>2</mml:mn></mml:mfrac><mml:mrow><mml:msub><mml:mi>k</mml:mi><mml:mrow><mml:mi>G</mml:mi><mml:mi>C</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msup><mml:msup><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>−</mml:mo><mml:mfrac><mml:msub><mml:mi>g</mml:mi><mml:mrow><mml:mi>r</mml:mi></mml:mrow></mml:msub><mml:mn>2</mml:mn></mml:mfrac></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mrow><mml:msub><mml:mi>k</mml:mi><mml:mrow><mml:mi>A</mml:mi><mml:mi>T</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msup><mml:mo>,</mml:mo></mml:mrow></mml:math></disp-formula></p><p>where <inline-formula><mml:math id="inf28"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:msub><mml:mi>k</mml:mi><mml:mrow><mml:mi>G</mml:mi><mml:mi>C</mml:mi></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mi>k</mml:mi><mml:mrow><mml:mi>A</mml:mi><mml:mi>T</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mn>3</mml:mn></mml:mrow></mml:mstyle></mml:math></inline-formula> total positions in codon <inline-formula><mml:math id="inf29"><mml:mi>i</mml:mi></mml:math></inline-formula>. The expected probability that amino acid <inline-formula><mml:math id="inf30"><mml:mi>a</mml:mi></mml:math></inline-formula> in region <inline-formula><mml:math id="inf31"><mml:mi>r</mml:mi></mml:math></inline-formula> is encoded by codon <inline-formula><mml:math id="inf32"><mml:mi>i</mml:mi></mml:math></inline-formula> is<disp-formula id="equ6"><label>(6)</label><mml:math id="m6"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:msub><mml:mi>E</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>r</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mfrac><mml:msub><mml:mi>p</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>r</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:munderover><mml:mo>∑</mml:mo><mml:mrow><mml:mi>j</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:msub><mml:mi>n</mml:mi><mml:mrow><mml:mi>a</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:munderover><mml:msub><mml:mi>p</mml:mi><mml:mrow><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>r</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfrac><mml:mo>.</mml:mo></mml:mstyle></mml:mrow></mml:math></disp-formula></p><p>We can then measure the degree to which the observed codon frequencies diverge from these expected probabilities using the Kullback–Leibler divergence. This gives a CAIS metric for a species <inline-formula><mml:math id="inf33"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>s</mml:mi></mml:mrow></mml:mstyle></mml:math></inline-formula> where <inline-formula><mml:math id="inf34"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:msub><mml:mi>O</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>s</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mstyle></mml:math></inline-formula> is the observed frequency of codon <italic>i</italic>:<disp-formula id="equ7"><label>(7)</label><mml:math id="m7"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mi>C</mml:mi><mml:mi>A</mml:mi><mml:mi>I</mml:mi><mml:mi>S</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mi>s</mml:mi><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:msubsup><mml:mi mathvariant="normal">Σ</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mn>64</mml:mn></mml:mrow></mml:msubsup><mml:msub><mml:mi>O</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>s</mml:mi></mml:mrow></mml:msub><mml:mspace width="thinmathspace"/><mml:mrow><mml:mi mathvariant="normal">l</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">o</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">g</mml:mi></mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mfrac><mml:msub><mml:mi>O</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>s</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mi>E</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>s</mml:mi></mml:mrow></mml:msub></mml:mfrac><mml:mo>)</mml:mo></mml:mrow><mml:mo>.</mml:mo></mml:mstyle></mml:mrow></mml:math></disp-formula></p></sec><sec id="s4-4-2"><title>Controlling for amino acid composition</title><p>Some amino acids may be more intrinsically prone to codon bias. We want a metric that quantifies effectiveness of selection (not amino acid frequency), so we re-weight CAIS on the basis of a standardized amino acid composition, to remove the effect of variation among species in amino acid frequencies.</p><p>Let <inline-formula><mml:math id="inf35"><mml:msub><mml:mrow><mml:mi>F</mml:mi></mml:mrow><mml:mrow><mml:mi>a</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> be the frequency of amino acid <inline-formula><mml:math id="inf36"><mml:mi>a</mml:mi></mml:math></inline-formula> across the entire dataset of 118 vertebrate genomes. We want to re-weight <inline-formula><mml:math id="inf37"><mml:msub><mml:mrow><mml:mi>O</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>s</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> on the basis of <inline-formula><mml:math id="inf38"><mml:msub><mml:mrow><mml:mi>F</mml:mi></mml:mrow><mml:mrow><mml:mi>a</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> to ensure that differences in amino acid frequencies among species do not affect CAIS, while preserving relative codon frequencies for the same amino acid. We do this by solving for <inline-formula><mml:math id="inf39"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:msub><mml:mi>α</mml:mi><mml:mrow><mml:mi>a</mml:mi><mml:mo>,</mml:mo><mml:mi>s</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mstyle></mml:math></inline-formula> so that<disp-formula id="equ8"><label>(8)</label><mml:math id="m8"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:msub><mml:mi>F</mml:mi><mml:mrow><mml:mi>a</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mi>α</mml:mi><mml:mrow><mml:mi>a</mml:mi><mml:mo>,</mml:mo><mml:mi>s</mml:mi></mml:mrow></mml:msub><mml:msubsup><mml:mi mathvariant="normal">Σ</mml:mi><mml:mrow><mml:mi>j</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:msub><mml:mi>n</mml:mi><mml:mrow><mml:mi>a</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msubsup><mml:msub><mml:mi>O</mml:mi><mml:mrow><mml:mi>j</mml:mi><mml:mo>,</mml:mo><mml:mi>s</mml:mi></mml:mrow></mml:msub><mml:mo>.</mml:mo></mml:mstyle></mml:mrow></mml:math></disp-formula></p><p>We then define <inline-formula><mml:math id="inf40"><mml:msubsup><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>s</mml:mi></mml:mrow><mml:mrow><mml:mi>`</mml:mi></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:msub><mml:mrow><mml:mi>α</mml:mi></mml:mrow><mml:mrow><mml:mi>a</mml:mi><mml:mo>,</mml:mo><mml:mi>s</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mrow><mml:mi>O</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>s</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> to obtain an amino acid frequency adjusted CAIS:<disp-formula id="equ9"><label>(9)</label><mml:math id="m9"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mi>C</mml:mi><mml:mi>A</mml:mi><mml:mi>I</mml:mi><mml:mi>S</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mi>S</mml:mi><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:msubsup><mml:mi mathvariant="normal">Σ</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mn>64</mml:mn></mml:mrow></mml:msubsup><mml:mspace width="thinmathspace"/><mml:msubsup><mml:mi>f</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>s</mml:mi></mml:mrow><mml:mrow><mml:mo>′</mml:mo></mml:mrow></mml:msubsup><mml:mspace width="thinmathspace"/><mml:mi>log</mml:mi><mml:mo>⁡</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mfrac><mml:msub><mml:mi>O</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>s</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mi>E</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>s</mml:mi></mml:mrow></mml:msub></mml:mfrac><mml:mo>)</mml:mo></mml:mrow><mml:mo>.</mml:mo></mml:mstyle></mml:mrow></mml:math></disp-formula></p><p>The <inline-formula><mml:math id="inf41"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:msub><mml:mi>F</mml:mi><mml:mrow><mml:mi>a</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mstyle></mml:math></inline-formula> values for our species set are at <ext-link ext-link-type="uri" xlink:href="https://github.com/MaselLab/Codon-Adaptation-Index-of-Species/blob/main/CAIS_ENC_calculation/Total_amino_acid_frequency_vertebrates.txt">https://github.com/MaselLab/Codon-Adaptation-Index-of-Species/blob/main/CAIS_ENC_calculation/Total_amino_acid_frequency_vertebrates.txt</ext-link>. Use of the standardized set of amino acid frequencies <inline-formula><mml:math id="inf42"><mml:msub><mml:mrow><mml:mi>F</mml:mi></mml:mrow><mml:mrow><mml:mi>a</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> has only a small effect on computed CAIS values relative to using each vertebrate species’ own amino acid frequencies (<xref ref-type="fig" rid="fig3s3">Figure 3—figure supplement 3</xref>).</p><p>CAIS corrected for local intergenic GC content but not species-wide amino acid composition is<disp-formula id="equ10"><label>(10)</label><mml:math id="m10"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mi>C</mml:mi><mml:mi>A</mml:mi><mml:mi>I</mml:mi><mml:msub><mml:mi>S</mml:mi><mml:mrow><mml:mi>l</mml:mi><mml:mi>o</mml:mi><mml:mi>c</mml:mi><mml:mi>a</mml:mi><mml:mi>l</mml:mi><mml:mi>G</mml:mi><mml:mi>C</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo>(</mml:mo><mml:mi>s</mml:mi><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:msup><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msubsup><mml:mi mathvariant="normal">Π</mml:mi><mml:mrow><mml:mi>r</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>G</mml:mi></mml:mrow></mml:msubsup><mml:msubsup><mml:mi mathvariant="normal">Σ</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mn>64</mml:mn></mml:mrow></mml:msubsup><mml:msub><mml:mi>O</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>r</mml:mi></mml:mrow></mml:msub><mml:mi>log</mml:mi><mml:mo>⁡</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mfrac><mml:msub><mml:mi>O</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>r</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mi>E</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>r</mml:mi></mml:mrow></mml:msub></mml:mfrac><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mrow><mml:mfrac><mml:mn>1</mml:mn><mml:mi>L</mml:mi></mml:mfrac></mml:mrow></mml:msup><mml:mo>,</mml:mo></mml:mstyle></mml:mrow></mml:math></disp-formula></p><p>where <inline-formula><mml:math id="inf43"><mml:msub><mml:mrow><mml:mi>O</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>r</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> is the number of times codon <inline-formula><mml:math id="inf44"><mml:mi>i</mml:mi></mml:math></inline-formula> appears in region <inline-formula><mml:math id="inf45"><mml:mi>r</mml:mi></mml:math></inline-formula> of species <inline-formula><mml:math id="inf46"><mml:mi>s</mml:mi></mml:math></inline-formula>, <inline-formula><mml:math id="inf47"><mml:msub><mml:mrow><mml:mi>E</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>r</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> is the expected number of times codon <inline-formula><mml:math id="inf48"><mml:mi>i</mml:mi></mml:math></inline-formula> would appear in region <inline-formula><mml:math id="inf49"><mml:mi>r</mml:mi></mml:math></inline-formula> of species <inline-formula><mml:math id="inf50"><mml:mi>s</mml:mi></mml:math></inline-formula> given the local intergenic GC content, <inline-formula><mml:math id="inf51"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>G</mml:mi></mml:mrow></mml:mstyle></mml:math></inline-formula> is the number of regions, and <inline-formula><mml:math id="inf52"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>L</mml:mi><mml:mo>=</mml:mo><mml:munderover><mml:mo>∑</mml:mo><mml:mrow><mml:mi>r</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>G</mml:mi></mml:mrow></mml:munderover><mml:munderover><mml:mo>∑</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mn>64</mml:mn></mml:mrow></mml:munderover><mml:msub><mml:mi>O</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>r</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mstyle></mml:math></inline-formula> is the total number of codons in the genome. Rewritten for greater computational ease:<disp-formula id="equ11"><label>(11)</label><mml:math id="m11"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mi>C</mml:mi><mml:mi>A</mml:mi><mml:mi>I</mml:mi><mml:msub><mml:mi>S</mml:mi><mml:mrow><mml:mi>l</mml:mi><mml:mi>o</mml:mi><mml:mi>c</mml:mi><mml:mi>a</mml:mi><mml:mi>l</mml:mi><mml:mi>G</mml:mi><mml:mi>C</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo>(</mml:mo><mml:mi>s</mml:mi><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:msup><mml:mi>e</mml:mi><mml:mrow><mml:mfrac><mml:mn>1</mml:mn><mml:mi>L</mml:mi></mml:mfrac><mml:msubsup><mml:mi mathvariant="normal">Σ</mml:mi><mml:mrow><mml:mi>r</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>G</mml:mi></mml:mrow></mml:msubsup><mml:mspace width="thinmathspace"/><mml:mrow><mml:mi mathvariant="normal">l</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">n</mml:mi></mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:munderover><mml:mo>∑</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mn>64</mml:mn></mml:mrow></mml:munderover><mml:msub><mml:mi>O</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>r</mml:mi></mml:mrow></mml:msub><mml:mi>log</mml:mi><mml:mo>⁡</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mfrac><mml:msub><mml:mi>O</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>r</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mi>E</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>r</mml:mi></mml:mrow></mml:msub></mml:mfrac><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:msup><mml:mo>.</mml:mo></mml:mstyle></mml:mrow></mml:math></disp-formula></p><p>Given the limited impact of amino acid frequency correction, we used <xref ref-type="disp-formula" rid="equ11">Equation 11</xref> for the local GC results, but we could correct for amino acid composition by replacing the <inline-formula><mml:math id="inf53"><mml:msub><mml:mrow><mml:mi>O</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>r</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> prefactor with <inline-formula><mml:math id="inf54"><mml:msubsup><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>s</mml:mi></mml:mrow><mml:mrow><mml:mi>`</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula>, or even <inline-formula><mml:math id="inf55"><mml:msubsup><mml:mrow><mml:mi>f</mml:mi></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>r</mml:mi></mml:mrow><mml:mrow><mml:mi>`</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula>.</p></sec></sec><sec id="s4-5"><title>Novembre’s ENC controlled for total GC content</title><p>The expected number of codons is based on the squared deviations <inline-formula><mml:math id="inf56"><mml:msubsup><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mi>a</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msubsup></mml:math></inline-formula> of the frequencies of the codons for each amino acid <inline-formula><mml:math id="inf57"><mml:mi>a</mml:mi></mml:math></inline-formula> from null expectations:<disp-formula id="equ12"> <label> (12)</label><mml:math id="m12"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:msubsup><mml:mi>X</mml:mi><mml:mrow><mml:mi>a</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:msubsup><mml:mi mathvariant="normal">Σ</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:msub><mml:mi>n</mml:mi><mml:mrow><mml:mi>a</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msubsup><mml:mfrac><mml:mrow><mml:msub><mml:mi>N</mml:mi><mml:mrow><mml:mi>a</mml:mi></mml:mrow></mml:msub><mml:msup><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mi>O</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>−</mml:mo><mml:msub><mml:mi>E</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup></mml:mrow><mml:msub><mml:mi>E</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mfrac><mml:mo>,</mml:mo></mml:mstyle></mml:mrow></mml:math></disp-formula></p><p>where <inline-formula><mml:math id="inf58"><mml:msub><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>a</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> is the total number of times that amino acid <inline-formula><mml:math id="inf59"><mml:mi>a</mml:mi></mml:math></inline-formula> appears. <xref ref-type="bibr" rid="bib50">Novembre, 2002</xref> defines the corrected ‘<italic>F</italic> value’ of amino acid <inline-formula><mml:math id="inf60"><mml:mi>a</mml:mi></mml:math></inline-formula> as<disp-formula id="equ13"><label>(13)</label><mml:math id="m13"><mml:mrow><mml:msub><mml:mrow><mml:mover><mml:mrow><mml:msup><mml:mi>F</mml:mi><mml:mo>′</mml:mo></mml:msup></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>a</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:msubsup><mml:mi>X</mml:mi><mml:mrow><mml:mi>a</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msubsup><mml:mo>+</mml:mo><mml:msub><mml:mi>N</mml:mi><mml:mrow><mml:mi>a</mml:mi></mml:mrow></mml:msub><mml:mo>−</mml:mo><mml:msub><mml:mi>n</mml:mi><mml:mrow><mml:mi>a</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:msub><mml:mi>n</mml:mi><mml:mrow><mml:mi>a</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mi>N</mml:mi><mml:mrow><mml:mi>a</mml:mi></mml:mrow></mml:msub><mml:mo>−</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:mfrac></mml:mrow></mml:math></disp-formula></p><p>and<disp-formula id="equ14"><label>(14)</label><mml:math id="m14"><mml:mrow><mml:mi>E</mml:mi><mml:mi>N</mml:mi><mml:mi>C</mml:mi><mml:mo>=</mml:mo><mml:mn>2</mml:mn><mml:mo>+</mml:mo><mml:mspace width="thinmathspace"/><mml:mfrac><mml:mn>9</mml:mn><mml:msub><mml:mrow><mml:mover><mml:msup><mml:mi>F</mml:mi><mml:mo>′</mml:mo></mml:msup><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:mfrac><mml:mo>+</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:msub><mml:mrow><mml:mover><mml:msup><mml:mi>F</mml:mi><mml:mo>′</mml:mo></mml:msup><mml:mo stretchy="false">^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mn>3</mml:mn></mml:mrow></mml:msub></mml:mfrac><mml:mtext>+</mml:mtext><mml:mfrac><mml:mn>5</mml:mn><mml:msub><mml:mrow><mml:mover><mml:msup><mml:mi>F</mml:mi><mml:mo>′</mml:mo></mml:msup><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mn>4</mml:mn></mml:mrow></mml:msub></mml:mfrac><mml:mo>+</mml:mo><mml:mfrac><mml:mn>3</mml:mn><mml:msub><mml:mrow><mml:mover><mml:msup><mml:mi>F</mml:mi><mml:mo>′</mml:mo></mml:msup><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mn>6</mml:mn></mml:mrow></mml:msub></mml:mfrac><mml:mo>,</mml:mo></mml:mrow></mml:math></disp-formula></p><p>where each <inline-formula><mml:math id="inf61"><mml:msub><mml:mrow><mml:mover accent="true"><mml:mrow><mml:mi>F</mml:mi><mml:mi>`</mml:mi></mml:mrow><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mi>n</mml:mi></mml:mrow><mml:mrow><mml:mi>a</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msub></mml:math></inline-formula> is the average of the ‘<italic>F</italic> values’ for amino acids with <inline-formula><mml:math id="inf62"><mml:msub><mml:mrow><mml:mi>n</mml:mi></mml:mrow><mml:mrow><mml:mi>a</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> synonymous codons. Past measures of ENC do not contain stop or start codons (<xref ref-type="bibr" rid="bib74">Wright, 1990</xref>; <xref ref-type="bibr" rid="bib50">Novembre, 2002</xref>; <xref ref-type="bibr" rid="bib23">Fuglsang, 2004</xref>), but as we did for CAI and CAIS above, we include stop codons as an ‘amino acid’ and therefore amend <xref ref-type="disp-formula" rid="equ14">Equation 14</xref> to<disp-formula id="equ15"><label>(15)</label><mml:math id="m15"><mml:mrow><mml:mi>E</mml:mi><mml:mi>N</mml:mi><mml:mi>C</mml:mi><mml:mo>=</mml:mo><mml:mn>2</mml:mn><mml:mo>+</mml:mo><mml:mspace width="thinmathspace"/><mml:mfrac><mml:mn>9</mml:mn><mml:msub><mml:mrow><mml:mover><mml:msup><mml:mi>F</mml:mi><mml:mo>′</mml:mo></mml:msup><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msub></mml:mfrac><mml:mo>+</mml:mo><mml:mfrac><mml:mn>2</mml:mn><mml:msub><mml:mrow><mml:mover><mml:msup><mml:mi>F</mml:mi><mml:mo>′</mml:mo></mml:msup><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mn>3</mml:mn></mml:mrow></mml:msub></mml:mfrac><mml:mtext>+</mml:mtext><mml:mfrac><mml:mn>5</mml:mn><mml:msub><mml:mrow><mml:mover><mml:msup><mml:mi>F</mml:mi><mml:mo>′</mml:mo></mml:msup><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mn>4</mml:mn></mml:mrow></mml:msub></mml:mfrac><mml:mo>+</mml:mo><mml:mfrac><mml:mn>3</mml:mn><mml:msub><mml:mrow><mml:mover><mml:msup><mml:mi>F</mml:mi><mml:mo>′</mml:mo></mml:msup><mml:mo>^</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mn>6</mml:mn></mml:mrow></mml:msub></mml:mfrac><mml:mo>.</mml:mo></mml:mrow></mml:math></disp-formula></p></sec><sec id="s4-6"><title>Statistical analysis</title><p>All statistical modeling was done in R 3.5.1. Scripts for calculating CAI and CAIS were written in Python 3.7.</p><sec id="s4-6-1"><title>Phylogenetic Independent Contrasts</title><p>Spurious phylogenetically confounded correlations can occur when closely related species share similar values of both metrics. One danger of such pseudoreplication is Simpson’s paradox, where there are negative slopes within taxonomic groups, but a positive slope among them might combine to yield an overall positive slope. We avoid pseudoreplication by using Phylogenetic Independent Contrasts (PIC) (<xref ref-type="bibr" rid="bib19">Felsenstein, 1985</xref>) to assess correlation. PIC analysis was done using the R package ‘ape’ (<xref ref-type="bibr" rid="bib55">Paradis and Schliep, 2019</xref>).</p></sec></sec></sec></body><back><sec sec-type="additional-information" id="s5"><title>Additional information</title><fn-group content-type="competing-interest"><title>Competing interests</title><fn fn-type="COI-statement" id="conf1"><p>No competing interests declared</p></fn></fn-group><fn-group content-type="author-contribution"><title>Author contributions</title><fn fn-type="con" id="con1"><p>Conceptualization, Data curation, Formal analysis, Funding acquisition, Investigation, Visualization, Methodology, Writing – original draft, Writing – review and editing</p></fn><fn fn-type="con" id="con2"><p>Formal analysis, Investigation, Visualization, Methodology, Writing – review and editing</p></fn><fn fn-type="con" id="con3"><p>Resources, Data curation, Supervision, Investigation, Methodology, Writing – original draft</p></fn><fn fn-type="con" id="con4"><p>Resources, Data curation, Supervision</p></fn><fn fn-type="con" id="con5"><p>Methodology, Writing – review and editing</p></fn><fn fn-type="con" id="con6"><p>Conceptualization, Formal analysis, Supervision, Funding acquisition, Methodology, Writing – original draft, Project administration, Writing – review and editing</p></fn></fn-group></sec><sec sec-type="supplementary-material" id="s6"><title>Additional files</title><supplementary-material id="mdar"><label>MDAR checklist</label><media xlink:href="elife-87335-mdarchecklist1-v1.pdf" mimetype="application" mime-subtype="pdf"/></supplementary-material></sec><sec sec-type="data-availability" id="s7"><title>Data availability</title><p>There is no new data. Processed data and code underlying this article are available in the public repository at <ext-link ext-link-type="uri" xlink:href="https://github.com/MaselLab/Codon-Adaptation-Index-of-Species">https://github.com/MaselLab/Codon-Adaptation-Index-of-Species</ext-link> (copy archived at <xref ref-type="bibr" rid="bib48">MaselLab, 2024</xref>).</p><p>The following previously published dataset was used:</p><p><element-citation publication-type="data" specific-use="references" id="dataset1"><person-group person-group-type="author"><name><surname>Jennifer</surname><given-names>J</given-names></name><name><surname>Sara</surname><given-names>W</given-names></name><name><surname>Paul</surname><given-names>N</given-names></name><name><surname>Catherine</surname><given-names>W</given-names></name><name><surname>Luke</surname><given-names>K</given-names></name><name><surname>Joanna</surname><given-names>M</given-names></name></person-group><year iso-8601-date="2020">2020</year><data-title>Data from: Universal and taxon-specific trends in protein sequences as a function of age</data-title><source>figshare</source><pub-id pub-id-type="doi">10.6084/m9.figshare.12037281.v1</pub-id></element-citation></p></sec><ack id="ack"><title>Acknowledgements</title><p>We thank Luke Kosinski, David Liberles, and Sawsan Wehbi for helpful discussions, Paul Nelson for providing the genome-wide GC contents, and the University of Arizona Undergraduate Biology Research Program for training. We thank Gavin Douglas for writing a convenient end-to-end implementation of CAIS on the basis of our preprint, which can be found at <ext-link ext-link-type="uri" xlink:href="https://github.com/gavinmdouglas/handy_pop_gen/blob/main/CAIS.py">https://github.com/gavinmdouglas/handy_pop_gen/blob/main/CAIS.py</ext-link>, and for catching a minor bug in our code in time for us to correct it in the version of record. We thank the anonymous reviewers for constructive feedback, and Laurent Duret for helpful elaboration on the concerns of reviewer 1.</p></ack><ref-list><title>References</title><ref id="bib1"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ahrens</surname><given-names>JB</given-names></name><name><surname>Nunez-Castilla</surname><given-names>J</given-names></name><name><surname>Siltberg-Liberles</surname><given-names>J</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>Evolution of intrinsic disorder in eukaryotic proteins</article-title><source>Cellular and Molecular Life Sciences</source><volume>74</volume><fpage>3163</fpage><lpage>3174</lpage><pub-id pub-id-type="doi">10.1007/s00018-017-2559-0</pub-id><pub-id pub-id-type="pmid">28597295</pub-id></element-citation></ref><ref id="bib2"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Akashi</surname><given-names>H</given-names></name></person-group><year iso-8601-date="1996">1996</year><article-title>Molecular evolution between <italic>Drosophila melanogaster</italic> and <italic>D. simulans</italic>: reduced codon bias, faster rates of amino acid substitution, and larger proteins in <italic>D. melanogaster</italic></article-title><source>Genetics</source><volume>144</volume><fpage>1297</fpage><lpage>1307</lpage><pub-id pub-id-type="doi">10.1093/genetics/144.3.1297</pub-id><pub-id pub-id-type="pmid">8913769</pub-id></element-citation></ref><ref id="bib3"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Basile</surname><given-names>W</given-names></name><name><surname>Salvatore</surname><given-names>M</given-names></name><name><surname>Bassot</surname><given-names>C</given-names></name><name><surname>Elofsson</surname><given-names>A</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Why do eukaryotic proteins contain more intrinsically disordered regions?</article-title><source>PLOS Computational Biology</source><volume>15</volume><elocation-id>e1007186</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pcbi.1007186</pub-id><pub-id pub-id-type="pmid">31329574</pub-id></element-citation></ref><ref id="bib4"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Bernardi</surname><given-names>G</given-names></name></person-group><year iso-8601-date="2000">2000</year><article-title>Isochores and the evolutionary genomics of vertebrates</article-title><source>Gene</source><volume>241</volume><fpage>3</fpage><lpage>17</lpage><pub-id pub-id-type="doi">10.1016/s0378-1119(99)00485-0</pub-id><pub-id pub-id-type="pmid">10607893</pub-id></element-citation></ref><ref id="bib5"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Bertram</surname><given-names>J</given-names></name><name><surname>Masel</surname><given-names>J</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>Evolution rapidly optimizes stability and aggregation in lattice proteins despite pervasive landscape valleys and mazes</article-title><source>Genetics</source><volume>214</volume><fpage>1047</fpage><lpage>1057</lpage><pub-id pub-id-type="doi">10.1534/genetics.120.302815</pub-id><pub-id pub-id-type="pmid">32107278</pub-id></element-citation></ref><ref id="bib6"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Botzman</surname><given-names>M</given-names></name><name><surname>Margalit</surname><given-names>H</given-names></name></person-group><year iso-8601-date="2011">2011</year><article-title>Variation in global codon usage bias among prokaryotic organisms is associated with their lifestyles</article-title><source>Genome Biology</source><volume>12</volume><elocation-id>R109</elocation-id><pub-id pub-id-type="doi">10.1186/gb-2011-12-10-r109</pub-id><pub-id pub-id-type="pmid">22032172</pub-id></element-citation></ref><ref id="bib7"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Brbić</surname><given-names>M</given-names></name><name><surname>Warnecke</surname><given-names>T</given-names></name><name><surname>Kriško</surname><given-names>A</given-names></name><name><surname>Supek</surname><given-names>F</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>Global shifts in genome and proteome composition are very tightly coupled</article-title><source>Genome Biology and Evolution</source><volume>7</volume><fpage>1519</fpage><lpage>1532</lpage><pub-id pub-id-type="doi">10.1093/gbe/evv088</pub-id><pub-id pub-id-type="pmid">25971281</pub-id></element-citation></ref><ref id="bib8"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Bulmer</surname><given-names>M</given-names></name></person-group><year iso-8601-date="1991">1991</year><article-title>The selection-mutation-drift theory of synonymous codon usage</article-title><source>Genetics</source><volume>129</volume><fpage>897</fpage><lpage>907</lpage><pub-id pub-id-type="doi">10.1093/genetics/129.3.897</pub-id><pub-id pub-id-type="pmid">1752426</pub-id></element-citation></ref><ref id="bib9"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Charlesworth</surname><given-names>B</given-names></name></person-group><year iso-8601-date="2009">2009</year><article-title>Fundamental concepts in genetics: effective population size and patterns of molecular evolution and variation</article-title><source>Nature Reviews. Genetics</source><volume>10</volume><fpage>195</fpage><lpage>205</lpage><pub-id pub-id-type="doi">10.1038/nrg2526</pub-id><pub-id pub-id-type="pmid">19204717</pub-id></element-citation></ref><ref id="bib10"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Cope</surname><given-names>AL</given-names></name><name><surname>Hettich</surname><given-names>RL</given-names></name><name><surname>Gilchrist</surname><given-names>MA</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Quantifying codon usage in signal peptides: Gene expression and amino acid usage explain apparent selection for inefficient codons</article-title><source>Biochimica et Biophysica Acta. Biomembranes</source><volume>1860</volume><fpage>2479</fpage><lpage>2485</lpage><pub-id pub-id-type="doi">10.1016/j.bbamem.2018.09.010</pub-id><pub-id pub-id-type="pmid">30279149</pub-id></element-citation></ref><ref id="bib11"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Daubin</surname><given-names>V</given-names></name><name><surname>Moran</surname><given-names>NA</given-names></name></person-group><year iso-8601-date="2004">2004</year><article-title>Comment on “The origins of genome complexity.”</article-title><source>Science</source><volume>306</volume><elocation-id>978</elocation-id><pub-id pub-id-type="doi">10.1126/science.1100559</pub-id><pub-id pub-id-type="pmid">15528429</pub-id></element-citation></ref><ref id="bib12"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Doherty</surname><given-names>A</given-names></name><name><surname>McInerney</surname><given-names>JO</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>Translational selection frequently overcomes genetic drift in shaping synonymous codon usage patterns in vertebrates</article-title><source>Molecular Biology and Evolution</source><volume>30</volume><fpage>2263</fpage><lpage>2267</lpage><pub-id pub-id-type="doi">10.1093/molbev/mst128</pub-id><pub-id pub-id-type="pmid">23883522</pub-id></element-citation></ref><ref id="bib13"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>dos Reis</surname><given-names>M</given-names></name><name><surname>Wernisch</surname><given-names>L</given-names></name></person-group><year iso-8601-date="2009">2009</year><article-title>Estimating translational selection in eukaryotic genomes</article-title><source>Molecular Biology and Evolution</source><volume>26</volume><fpage>451</fpage><lpage>461</lpage><pub-id pub-id-type="doi">10.1093/molbev/msn272</pub-id><pub-id pub-id-type="pmid">19033257</pub-id></element-citation></ref><ref id="bib14"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Doyle</surname><given-names>JM</given-names></name><name><surname>Hacking</surname><given-names>CC</given-names></name><name><surname>Willoughby</surname><given-names>JR</given-names></name><name><surname>Sundaram</surname><given-names>M</given-names></name><name><surname>DeWoody</surname><given-names>JA</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>Mammalian genetic diversity as a function of habitat, body size, trophic class, and conservation status</article-title><source>Journal of Mammalogy</source><volume>96</volume><fpage>564</fpage><lpage>572</lpage><pub-id pub-id-type="doi">10.1093/jmammal/gyv061</pub-id></element-citation></ref><ref id="bib15"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Dunker</surname><given-names>AK</given-names></name><name><surname>Garner</surname><given-names>E</given-names></name><name><surname>Guilliot</surname><given-names>S</given-names></name><name><surname>Romero</surname><given-names>P</given-names></name><name><surname>Albrecht</surname><given-names>K</given-names></name><name><surname>Hart</surname><given-names>J</given-names></name><name><surname>Obradovic</surname><given-names>Z</given-names></name><name><surname>Kissinger</surname><given-names>C</given-names></name><name><surname>Villafranca</surname><given-names>JE</given-names></name></person-group><year iso-8601-date="1998">1998</year><article-title>Protein Disorder and the Evolution of Molecular Recognition: Theory, Predictions and Observations</article-title><source>Pac Symp Biocomput</source><fpage>473</fpage><lpage>484</lpage></element-citation></ref><ref id="bib16"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Duret</surname><given-names>L</given-names></name><name><surname>Eyre-Walker</surname><given-names>A</given-names></name><name><surname>Galtier</surname><given-names>N</given-names></name></person-group><year iso-8601-date="2006">2006</year><article-title>A new perspective on isochore evolution</article-title><source>Gene</source><volume>385</volume><fpage>71</fpage><lpage>74</lpage><pub-id pub-id-type="doi">10.1016/j.gene.2006.04.030</pub-id><pub-id pub-id-type="pmid">16971063</pub-id></element-citation></ref><ref id="bib17"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Duret</surname><given-names>L</given-names></name><name><surname>Galtier</surname><given-names>N</given-names></name></person-group><year iso-8601-date="2009">2009</year><article-title>Biased gene conversion and the evolution of mammalian genomic landscapes</article-title><source>Annual Review of Genomics and Human Genetics</source><volume>10</volume><fpage>285</fpage><lpage>311</lpage><pub-id pub-id-type="doi">10.1146/annurev-genom-082908-150001</pub-id><pub-id pub-id-type="pmid">19630562</pub-id></element-citation></ref><ref id="bib18"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Eyre-Walker</surname><given-names>A</given-names></name><name><surname>Hurst</surname><given-names>LD</given-names></name></person-group><year iso-8601-date="2001">2001</year><article-title>The evolution of isochores</article-title><source>Nature Reviews. Genetics</source><volume>2</volume><fpage>549</fpage><lpage>555</lpage><pub-id pub-id-type="doi">10.1038/35080577</pub-id><pub-id pub-id-type="pmid">11433361</pub-id></element-citation></ref><ref id="bib19"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Felsenstein</surname><given-names>J</given-names></name></person-group><year iso-8601-date="1985">1985</year><article-title>Phylogenies and the comparative method</article-title><source>The American Naturalist</source><volume>125</volume><fpage>1</fpage><lpage>15</lpage><pub-id pub-id-type="doi">10.1086/284325</pub-id></element-citation></ref><ref id="bib20"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Figuet</surname><given-names>E</given-names></name><name><surname>Ballenghien</surname><given-names>M</given-names></name><name><surname>Romiguier</surname><given-names>J</given-names></name><name><surname>Galtier</surname><given-names>N</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>Biased gene conversion and GC-content evolution in the coding sequences of reptiles and vertebrates</article-title><source>Genome Biology and Evolution</source><volume>7</volume><fpage>240</fpage><lpage>250</lpage><pub-id pub-id-type="doi">10.1093/gbe/evu277</pub-id><pub-id pub-id-type="pmid">25527834</pub-id></element-citation></ref><ref id="bib21"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Forcelloni</surname><given-names>S</given-names></name><name><surname>Giansanti</surname><given-names>A</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>Evolutionary forces and codon bias in different flavors of intrinsic disorder in the human proteome</article-title><source>Journal of Molecular Evolution</source><volume>88</volume><fpage>164</fpage><lpage>178</lpage><pub-id pub-id-type="doi">10.1007/s00239-019-09921-4</pub-id><pub-id pub-id-type="pmid">31820049</pub-id></element-citation></ref><ref id="bib22"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Foy</surname><given-names>SG</given-names></name><name><surname>Wilson</surname><given-names>BA</given-names></name><name><surname>Bertram</surname><given-names>J</given-names></name><name><surname>Cordes</surname><given-names>MHJ</given-names></name><name><surname>Masel</surname><given-names>J</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>A shift in aggregation avoidance strategy marks a long-term direction to protein evolution</article-title><source>Genetics</source><volume>211</volume><fpage>1345</fpage><lpage>1355</lpage><pub-id pub-id-type="doi">10.1534/genetics.118.301719</pub-id><pub-id pub-id-type="pmid">30692195</pub-id></element-citation></ref><ref id="bib23"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Fuglsang</surname><given-names>A</given-names></name></person-group><year iso-8601-date="2004">2004</year><article-title>The ‘effective number of codons’ revisited</article-title><source>Biochemical and Biophysical Research Communications</source><volume>317</volume><fpage>957</fpage><lpage>964</lpage><pub-id pub-id-type="doi">10.1016/j.bbrc.2004.03.138</pub-id></element-citation></ref><ref id="bib24"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Fuglsang</surname><given-names>A</given-names></name></person-group><year iso-8601-date="2008">2008</year><article-title>Impact of bias discrepancy and amino acid usage on estimates of the effective number of codons used in a gene, and a test for selection on codon usage</article-title><source>Gene</source><volume>410</volume><fpage>82</fpage><lpage>88</lpage><pub-id pub-id-type="doi">10.1016/j.gene.2007.12.001</pub-id><pub-id pub-id-type="pmid">18248919</pub-id></element-citation></ref><ref id="bib25"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Galtier</surname><given-names>N</given-names></name><name><surname>Piganeau</surname><given-names>G</given-names></name><name><surname>Mouchiroud</surname><given-names>D</given-names></name><name><surname>Duret</surname><given-names>L</given-names></name></person-group><year iso-8601-date="2001">2001</year><article-title>GC-content evolution in mammalian genomes: the biased gene conversion hypothesis</article-title><source>Genetics</source><volume>159</volume><fpage>907</fpage><lpage>911</lpage><pub-id pub-id-type="doi">10.1093/genetics/159.2.907</pub-id><pub-id pub-id-type="pmid">11693127</pub-id></element-citation></ref><ref id="bib26"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Galtier</surname><given-names>N</given-names></name><name><surname>Roux</surname><given-names>C</given-names></name><name><surname>Rousselle</surname><given-names>M</given-names></name><name><surname>Romiguier</surname><given-names>J</given-names></name><name><surname>Figuet</surname><given-names>E</given-names></name><name><surname>Glémin</surname><given-names>S</given-names></name><name><surname>Bierne</surname><given-names>N</given-names></name><name><surname>Duret</surname><given-names>L</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Codon usage bias in animals: Disentangling the effects of natural selection, effective population size, and GC-biased gene conversion</article-title><source>Molecular Biology and Evolution</source><volume>35</volume><fpage>1092</fpage><lpage>1103</lpage><pub-id pub-id-type="doi">10.1093/molbev/msy015</pub-id></element-citation></ref><ref id="bib27"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Gingold</surname><given-names>H</given-names></name><name><surname>Tehler</surname><given-names>D</given-names></name><name><surname>Christoffersen</surname><given-names>NR</given-names></name><name><surname>Nielsen</surname><given-names>MM</given-names></name><name><surname>Asmar</surname><given-names>F</given-names></name><name><surname>Kooistra</surname><given-names>SM</given-names></name><name><surname>Christophersen</surname><given-names>NS</given-names></name><name><surname>Christensen</surname><given-names>LL</given-names></name><name><surname>Borre</surname><given-names>M</given-names></name><name><surname>Sørensen</surname><given-names>KD</given-names></name><name><surname>Andersen</surname><given-names>LD</given-names></name><name><surname>Andersen</surname><given-names>CL</given-names></name><name><surname>Hulleman</surname><given-names>E</given-names></name><name><surname>Wurdinger</surname><given-names>T</given-names></name><name><surname>Ralfkiær</surname><given-names>E</given-names></name><name><surname>Helin</surname><given-names>K</given-names></name><name><surname>Grønbæk</surname><given-names>K</given-names></name><name><surname>Ørntoft</surname><given-names>T</given-names></name><name><surname>Waszak</surname><given-names>SM</given-names></name><name><surname>Dahan</surname><given-names>O</given-names></name><name><surname>Pedersen</surname><given-names>JS</given-names></name><name><surname>Lund</surname><given-names>AH</given-names></name><name><surname>Pilpel</surname><given-names>Y</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>A dual program for translation regulation in cellular proliferation and differentiation</article-title><source>Cell</source><volume>158</volume><fpage>1281</fpage><lpage>1292</lpage><pub-id pub-id-type="doi">10.1016/j.cell.2014.08.011</pub-id><pub-id pub-id-type="pmid">25215487</pub-id></element-citation></ref><ref id="bib28"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Gossmann</surname><given-names>TI</given-names></name><name><surname>Keightley</surname><given-names>PD</given-names></name><name><surname>Eyre-Walker</surname><given-names>A</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>The effect of variation in the effective population size on the rate of adaptive molecular evolution in eukaryotes</article-title><source>Genome Biology and Evolution</source><volume>4</volume><fpage>658</fpage><lpage>667</lpage><pub-id pub-id-type="doi">10.1093/gbe/evs027</pub-id><pub-id pub-id-type="pmid">22436998</pub-id></element-citation></ref><ref id="bib29"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hershberg</surname><given-names>R</given-names></name><name><surname>Petrov</surname><given-names>DA</given-names></name></person-group><year iso-8601-date="2008">2008</year><article-title>Selection on codon bias</article-title><source>Annual Review of Genetics</source><volume>42</volume><fpage>287</fpage><lpage>299</lpage><pub-id pub-id-type="doi">10.1146/annurev.genet.42.110807.091442</pub-id><pub-id pub-id-type="pmid">18983258</pub-id></element-citation></ref><ref id="bib30"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hershberg</surname><given-names>R</given-names></name><name><surname>Petrov</surname><given-names>DA</given-names></name></person-group><year iso-8601-date="2009">2009</year><article-title>General rules for optimal codon choice</article-title><source>PLOS Genetics</source><volume>5</volume><elocation-id>e1000556</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pgen.1000556</pub-id><pub-id pub-id-type="pmid">19593368</pub-id></element-citation></ref><ref id="bib31"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hildebrand</surname><given-names>F</given-names></name><name><surname>Meyer</surname><given-names>A</given-names></name><name><surname>Eyre-Walker</surname><given-names>A</given-names></name></person-group><year iso-8601-date="2010">2010</year><article-title>Evidence of selection upon genomic GC-content in bacteria</article-title><source>PLOS Genetics</source><volume>6</volume><elocation-id>e1001107</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pgen.1001107</pub-id><pub-id pub-id-type="pmid">20838593</pub-id></element-citation></ref><ref id="bib32"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Huang</surname><given-names>Y</given-names></name><name><surname>Liu</surname><given-names>Z</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>Do intrinsically disordered proteins possess high specificity in protein–protein interactions?</article-title><source>Chemistry – A European Journal</source><volume>19</volume><fpage>4462</fpage><lpage>4467</lpage><pub-id pub-id-type="doi">10.1002/chem.201203100</pub-id></element-citation></ref><ref id="bib33"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hunt</surname><given-names>RC</given-names></name><name><surname>Simhadri</surname><given-names>VL</given-names></name><name><surname>Iandoli</surname><given-names>M</given-names></name><name><surname>Sauna</surname><given-names>ZE</given-names></name><name><surname>Kimchi-Sarfaty</surname><given-names>C</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>Exposing synonymous mutations</article-title><source>Trends in Genetics</source><volume>30</volume><fpage>308</fpage><lpage>321</lpage><pub-id pub-id-type="doi">10.1016/j.tig.2014.04.006</pub-id><pub-id pub-id-type="pmid">24954581</pub-id></element-citation></ref><ref id="bib34"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>James</surname><given-names>JE</given-names></name><name><surname>Willis</surname><given-names>SM</given-names></name><name><surname>Nelson</surname><given-names>PG</given-names></name><name><surname>Weibel</surname><given-names>C</given-names></name><name><surname>Kosinski</surname><given-names>LJ</given-names></name><name><surname>Masel</surname><given-names>J</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>Universal and taxon-specific trends in protein sequences as a function of age</article-title><source>eLife</source><volume>10</volume><elocation-id>e57347</elocation-id><pub-id pub-id-type="doi">10.7554/eLife.57347</pub-id><pub-id pub-id-type="pmid">33416492</pub-id></element-citation></ref><ref id="bib35"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>James</surname><given-names>JE</given-names></name><name><surname>Nelson</surname><given-names>PG</given-names></name><name><surname>Masel</surname><given-names>J</given-names></name></person-group><year iso-8601-date="2023">2023</year><article-title>Differential retention of pfam domains contributes to long-term evolutionary trends</article-title><source>Molecular Biology and Evolution</source><volume>40</volume><elocation-id>msad073</elocation-id><pub-id pub-id-type="doi">10.1093/molbev/msad073</pub-id><pub-id pub-id-type="pmid">36947137</pub-id></element-citation></ref><ref id="bib36"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kessler</surname><given-names>MD</given-names></name><name><surname>Dean</surname><given-names>MD</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>Effective population size does not predict codon usage bias in mammals</article-title><source>Ecology and Evolution</source><volume>4</volume><fpage>3887</fpage><lpage>3900</lpage><pub-id pub-id-type="doi">10.1002/ece3.1249</pub-id><pub-id pub-id-type="pmid">25505518</pub-id></element-citation></ref><ref id="bib37"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kimura</surname><given-names>M</given-names></name></person-group><year iso-8601-date="1962">1962</year><article-title>On the probability of fixation of mutant genes in a population</article-title><source>Genetics</source><volume>47</volume><fpage>713</fpage><lpage>719</lpage><pub-id pub-id-type="doi">10.1093/genetics/47.6.713</pub-id><pub-id pub-id-type="pmid">14456043</pub-id></element-citation></ref><ref id="bib38"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kondrashov</surname><given-names>AS</given-names></name><name><surname>Povolotskaya</surname><given-names>IS</given-names></name><name><surname>Ivankov</surname><given-names>DN</given-names></name><name><surname>Kondrashov</surname><given-names>FA</given-names></name></person-group><year iso-8601-date="2010">2010</year><article-title>Rate of sequence divergence under constant selection</article-title><source>Biology Direct</source><volume>5</volume><elocation-id>5</elocation-id><pub-id pub-id-type="doi">10.1186/1745-6150-5-5</pub-id><pub-id pub-id-type="pmid">20092641</pub-id></element-citation></ref><ref id="bib39"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kosinski</surname><given-names>LJ</given-names></name><name><surname>Aviles</surname><given-names>NR</given-names></name><name><surname>Gomez</surname><given-names>K</given-names></name><name><surname>Masel</surname><given-names>J</given-names></name></person-group><year iso-8601-date="2022">2022</year><article-title>Random peptides rich in small and disorder-promoting amino acids are less likely to be harmful</article-title><source>Genome Biology and Evolution</source><volume>14</volume><elocation-id>evac085</elocation-id><pub-id pub-id-type="doi">10.1093/gbe/evac085</pub-id><pub-id pub-id-type="pmid">35668555</pub-id></element-citation></ref><ref id="bib40"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kullback</surname><given-names>S</given-names></name><name><surname>Leibler</surname><given-names>RA</given-names></name></person-group><year iso-8601-date="1951">1951</year><article-title>On information and sufficiency</article-title><source>The Annals of Mathematical Statistics</source><volume>22</volume><fpage>79</fpage><lpage>86</lpage><pub-id pub-id-type="doi">10.1214/aoms/1177729694</pub-id></element-citation></ref><ref id="bib41"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kumar</surname><given-names>S</given-names></name><name><surname>Stecher</surname><given-names>G</given-names></name><name><surname>Suleski</surname><given-names>M</given-names></name><name><surname>Hedges</surname><given-names>SB</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>TimeTree: A resource for timelines, timetrees, and divergence times</article-title><source>Molecular Biology and Evolution</source><volume>34</volume><fpage>1812</fpage><lpage>1819</lpage><pub-id pub-id-type="doi">10.1093/molbev/msx116</pub-id><pub-id pub-id-type="pmid">28387841</pub-id></element-citation></ref><ref id="bib42"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Lander</surname><given-names>ES</given-names></name><name><surname>Linton</surname><given-names>LM</given-names></name><name><surname>Birren</surname><given-names>B</given-names></name><name><surname>Nusbaum</surname><given-names>C</given-names></name><name><surname>Zody</surname><given-names>MC</given-names></name><name><surname>Baldwin</surname><given-names>J</given-names></name><name><surname>Devon</surname><given-names>K</given-names></name><name><surname>Dewar</surname><given-names>K</given-names></name><name><surname>Doyle</surname><given-names>M</given-names></name><name><surname>FitzHugh</surname><given-names>W</given-names></name><name><surname>Funke</surname><given-names>R</given-names></name><name><surname>Gage</surname><given-names>D</given-names></name><name><surname>Harris</surname><given-names>K</given-names></name><name><surname>Heaford</surname><given-names>A</given-names></name><name><surname>Howland</surname><given-names>J</given-names></name><name><surname>Kann</surname><given-names>L</given-names></name><name><surname>Lehoczky</surname><given-names>J</given-names></name><name><surname>LeVine</surname><given-names>R</given-names></name><name><surname>McEwan</surname><given-names>P</given-names></name><name><surname>McKernan</surname><given-names>K</given-names></name><name><surname>Meldrim</surname><given-names>J</given-names></name><name><surname>Mesirov</surname><given-names>JP</given-names></name><name><surname>Miranda</surname><given-names>C</given-names></name><name><surname>Morris</surname><given-names>W</given-names></name><name><surname>Naylor</surname><given-names>J</given-names></name><name><surname>Raymond</surname><given-names>C</given-names></name><name><surname>Rosetti</surname><given-names>M</given-names></name><name><surname>Santos</surname><given-names>R</given-names></name><name><surname>Sheridan</surname><given-names>A</given-names></name><name><surname>Sougnez</surname><given-names>C</given-names></name><name><surname>Stange-Thomann</surname><given-names>Y</given-names></name><name><surname>Stojanovic</surname><given-names>N</given-names></name><name><surname>Subramanian</surname><given-names>A</given-names></name><name><surname>Wyman</surname><given-names>D</given-names></name><name><surname>Rogers</surname><given-names>J</given-names></name><name><surname>Sulston</surname><given-names>J</given-names></name><name><surname>Ainscough</surname><given-names>R</given-names></name><name><surname>Beck</surname><given-names>S</given-names></name><name><surname>Bentley</surname><given-names>D</given-names></name><name><surname>Burton</surname><given-names>J</given-names></name><name><surname>Clee</surname><given-names>C</given-names></name><name><surname>Carter</surname><given-names>N</given-names></name><name><surname>Coulson</surname><given-names>A</given-names></name><name><surname>Deadman</surname><given-names>R</given-names></name><name><surname>Deloukas</surname><given-names>P</given-names></name><name><surname>Dunham</surname><given-names>A</given-names></name><name><surname>Dunham</surname><given-names>I</given-names></name><name><surname>Durbin</surname><given-names>R</given-names></name><name><surname>French</surname><given-names>L</given-names></name><name><surname>Grafham</surname><given-names>D</given-names></name><name><surname>Gregory</surname><given-names>S</given-names></name><name><surname>Hubbard</surname><given-names>T</given-names></name><name><surname>Humphray</surname><given-names>S</given-names></name><name><surname>Hunt</surname><given-names>A</given-names></name><name><surname>Jones</surname><given-names>M</given-names></name><name><surname>Lloyd</surname><given-names>C</given-names></name><name><surname>McMurray</surname><given-names>A</given-names></name><name><surname>Matthews</surname><given-names>L</given-names></name><name><surname>Mercer</surname><given-names>S</given-names></name><name><surname>Milne</surname><given-names>S</given-names></name><name><surname>Mullikin</surname><given-names>JC</given-names></name><name><surname>Mungall</surname><given-names>A</given-names></name><name><surname>Plumb</surname><given-names>R</given-names></name><name><surname>Ross</surname><given-names>M</given-names></name><name><surname>Shownkeen</surname><given-names>R</given-names></name><name><surname>Sims</surname><given-names>S</given-names></name><name><surname>Waterston</surname><given-names>RH</given-names></name><name><surname>Wilson</surname><given-names>RK</given-names></name><name><surname>Hillier</surname><given-names>LW</given-names></name><name><surname>McPherson</surname><given-names>JD</given-names></name><name><surname>Marra</surname><given-names>MA</given-names></name><name><surname>Mardis</surname><given-names>ER</given-names></name><name><surname>Fulton</surname><given-names>LA</given-names></name><name><surname>Chinwalla</surname><given-names>AT</given-names></name><name><surname>Pepin</surname><given-names>KH</given-names></name><name><surname>Gish</surname><given-names>WR</given-names></name><name><surname>Chissoe</surname><given-names>SL</given-names></name><name><surname>Wendl</surname><given-names>MC</given-names></name><name><surname>Delehaunty</surname><given-names>KD</given-names></name><name><surname>Miner</surname><given-names>TL</given-names></name><name><surname>Delehaunty</surname><given-names>A</given-names></name><name><surname>Kramer</surname><given-names>JB</given-names></name><name><surname>Cook</surname><given-names>LL</given-names></name><name><surname>Fulton</surname><given-names>RS</given-names></name><name><surname>Johnson</surname><given-names>DL</given-names></name><name><surname>Minx</surname><given-names>PJ</given-names></name><name><surname>Clifton</surname><given-names>SW</given-names></name><name><surname>Hawkins</surname><given-names>T</given-names></name><name><surname>Branscomb</surname><given-names>E</given-names></name><name><surname>Predki</surname><given-names>P</given-names></name><name><surname>Richardson</surname><given-names>P</given-names></name><name><surname>Wenning</surname><given-names>S</given-names></name><name><surname>Slezak</surname><given-names>T</given-names></name><name><surname>Doggett</surname><given-names>N</given-names></name><name><surname>Cheng</surname><given-names>JF</given-names></name><name><surname>Olsen</surname><given-names>A</given-names></name><name><surname>Lucas</surname><given-names>S</given-names></name><name><surname>Elkin</surname><given-names>C</given-names></name><name><surname>Uberbacher</surname><given-names>E</given-names></name><name><surname>Frazier</surname><given-names>M</given-names></name><name><surname>Gibbs</surname><given-names>RA</given-names></name><name><surname>Muzny</surname><given-names>DM</given-names></name><name><surname>Scherer</surname><given-names>SE</given-names></name><name><surname>Bouck</surname><given-names>JB</given-names></name><name><surname>Sodergren</surname><given-names>EJ</given-names></name><name><surname>Worley</surname><given-names>KC</given-names></name><name><surname>Rives</surname><given-names>CM</given-names></name><name><surname>Gorrell</surname><given-names>JH</given-names></name><name><surname>Metzker</surname><given-names>ML</given-names></name><name><surname>Naylor</surname><given-names>SL</given-names></name><name><surname>Kucherlapati</surname><given-names>RS</given-names></name><name><surname>Nelson</surname><given-names>DL</given-names></name><name><surname>Weinstock</surname><given-names>GM</given-names></name><name><surname>Sakaki</surname><given-names>Y</given-names></name><name><surname>Fujiyama</surname><given-names>A</given-names></name><name><surname>Hattori</surname><given-names>M</given-names></name><name><surname>Yada</surname><given-names>T</given-names></name><name><surname>Toyoda</surname><given-names>A</given-names></name><name><surname>Itoh</surname><given-names>T</given-names></name><name><surname>Kawagoe</surname><given-names>C</given-names></name><name><surname>Watanabe</surname><given-names>H</given-names></name><name><surname>Totoki</surname><given-names>Y</given-names></name><name><surname>Taylor</surname><given-names>T</given-names></name><name><surname>Weissenbach</surname><given-names>J</given-names></name><name><surname>Heilig</surname><given-names>R</given-names></name><name><surname>Saurin</surname><given-names>W</given-names></name><name><surname>Artiguenave</surname><given-names>F</given-names></name><name><surname>Brottier</surname><given-names>P</given-names></name><name><surname>Bruls</surname><given-names>T</given-names></name><name><surname>Pelletier</surname><given-names>E</given-names></name><name><surname>Robert</surname><given-names>C</given-names></name><name><surname>Wincker</surname><given-names>P</given-names></name><name><surname>Smith</surname><given-names>DR</given-names></name><name><surname>Doucette-Stamm</surname><given-names>L</given-names></name><name><surname>Rubenfield</surname><given-names>M</given-names></name><name><surname>Weinstock</surname><given-names>K</given-names></name><name><surname>Lee</surname><given-names>HM</given-names></name><name><surname>Dubois</surname><given-names>J</given-names></name><name><surname>Rosenthal</surname><given-names>A</given-names></name><name><surname>Platzer</surname><given-names>M</given-names></name><name><surname>Nyakatura</surname><given-names>G</given-names></name><name><surname>Taudien</surname><given-names>S</given-names></name><name><surname>Rump</surname><given-names>A</given-names></name><name><surname>Yang</surname><given-names>H</given-names></name><name><surname>Yu</surname><given-names>J</given-names></name><name><surname>Wang</surname><given-names>J</given-names></name><name><surname>Huang</surname><given-names>G</given-names></name><name><surname>Gu</surname><given-names>J</given-names></name><name><surname>Hood</surname><given-names>L</given-names></name><name><surname>Rowen</surname><given-names>L</given-names></name><name><surname>Madan</surname><given-names>A</given-names></name><name><surname>Qin</surname><given-names>S</given-names></name><name><surname>Davis</surname><given-names>RW</given-names></name><name><surname>Federspiel</surname><given-names>NA</given-names></name><name><surname>Abola</surname><given-names>AP</given-names></name><name><surname>Proctor</surname><given-names>MJ</given-names></name><name><surname>Myers</surname><given-names>RM</given-names></name><name><surname>Schmutz</surname><given-names>J</given-names></name><name><surname>Dickson</surname><given-names>M</given-names></name><name><surname>Grimwood</surname><given-names>J</given-names></name><name><surname>Cox</surname><given-names>DR</given-names></name><name><surname>Olson</surname><given-names>MV</given-names></name><name><surname>Kaul</surname><given-names>R</given-names></name><name><surname>Raymond</surname><given-names>C</given-names></name><name><surname>Shimizu</surname><given-names>N</given-names></name><name><surname>Kawasaki</surname><given-names>K</given-names></name><name><surname>Minoshima</surname><given-names>S</given-names></name><name><surname>Evans</surname><given-names>GA</given-names></name><name><surname>Athanasiou</surname><given-names>M</given-names></name><name><surname>Schultz</surname><given-names>R</given-names></name><name><surname>Roe</surname><given-names>BA</given-names></name><name><surname>Chen</surname><given-names>F</given-names></name><name><surname>Pan</surname><given-names>H</given-names></name><name><surname>Ramser</surname><given-names>J</given-names></name><name><surname>Lehrach</surname><given-names>H</given-names></name><name><surname>Reinhardt</surname><given-names>R</given-names></name><name><surname>McCombie</surname><given-names>WR</given-names></name><name><surname>de la Bastide</surname><given-names>M</given-names></name><name><surname>Dedhia</surname><given-names>N</given-names></name><name><surname>Blöcker</surname><given-names>H</given-names></name><name><surname>Hornischer</surname><given-names>K</given-names></name><name><surname>Nordsiek</surname><given-names>G</given-names></name><name><surname>Agarwala</surname><given-names>R</given-names></name><name><surname>Aravind</surname><given-names>L</given-names></name><name><surname>Bailey</surname><given-names>JA</given-names></name><name><surname>Bateman</surname><given-names>A</given-names></name><name><surname>Batzoglou</surname><given-names>S</given-names></name><name><surname>Birney</surname><given-names>E</given-names></name><name><surname>Bork</surname><given-names>P</given-names></name><name><surname>Brown</surname><given-names>DG</given-names></name><name><surname>Burge</surname><given-names>CB</given-names></name><name><surname>Cerutti</surname><given-names>L</given-names></name><name><surname>Chen</surname><given-names>HC</given-names></name><name><surname>Church</surname><given-names>D</given-names></name><name><surname>Clamp</surname><given-names>M</given-names></name><name><surname>Copley</surname><given-names>RR</given-names></name><name><surname>Doerks</surname><given-names>T</given-names></name><name><surname>Eddy</surname><given-names>SR</given-names></name><name><surname>Eichler</surname><given-names>EE</given-names></name><name><surname>Furey</surname><given-names>TS</given-names></name><name><surname>Galagan</surname><given-names>J</given-names></name><name><surname>Gilbert</surname><given-names>JG</given-names></name><name><surname>Harmon</surname><given-names>C</given-names></name><name><surname>Hayashizaki</surname><given-names>Y</given-names></name><name><surname>Haussler</surname><given-names>D</given-names></name><name><surname>Hermjakob</surname><given-names>H</given-names></name><name><surname>Hokamp</surname><given-names>K</given-names></name><name><surname>Jang</surname><given-names>W</given-names></name><name><surname>Johnson</surname><given-names>LS</given-names></name><name><surname>Jones</surname><given-names>TA</given-names></name><name><surname>Kasif</surname><given-names>S</given-names></name><name><surname>Kaspryzk</surname><given-names>A</given-names></name><name><surname>Kennedy</surname><given-names>S</given-names></name><name><surname>Kent</surname><given-names>WJ</given-names></name><name><surname>Kitts</surname><given-names>P</given-names></name><name><surname>Koonin</surname><given-names>EV</given-names></name><name><surname>Korf</surname><given-names>I</given-names></name><name><surname>Kulp</surname><given-names>D</given-names></name><name><surname>Lancet</surname><given-names>D</given-names></name><name><surname>Lowe</surname><given-names>TM</given-names></name><name><surname>McLysaght</surname><given-names>A</given-names></name><name><surname>Mikkelsen</surname><given-names>T</given-names></name><name><surname>Moran</surname><given-names>JV</given-names></name><name><surname>Mulder</surname><given-names>N</given-names></name><name><surname>Pollara</surname><given-names>VJ</given-names></name><name><surname>Ponting</surname><given-names>CP</given-names></name><name><surname>Schuler</surname><given-names>G</given-names></name><name><surname>Schultz</surname><given-names>J</given-names></name><name><surname>Slater</surname><given-names>G</given-names></name><name><surname>Smit</surname><given-names>AF</given-names></name><name><surname>Stupka</surname><given-names>E</given-names></name><name><surname>Szustakowki</surname><given-names>J</given-names></name><name><surname>Thierry-Mieg</surname><given-names>D</given-names></name><name><surname>Thierry-Mieg</surname><given-names>J</given-names></name><name><surname>Wagner</surname><given-names>L</given-names></name><name><surname>Wallis</surname><given-names>J</given-names></name><name><surname>Wheeler</surname><given-names>R</given-names></name><name><surname>Williams</surname><given-names>A</given-names></name><name><surname>Wolf</surname><given-names>YI</given-names></name><name><surname>Wolfe</surname><given-names>KH</given-names></name><name><surname>Yang</surname><given-names>SP</given-names></name><name><surname>Yeh</surname><given-names>RF</given-names></name><name><surname>Collins</surname><given-names>F</given-names></name><name><surname>Guyer</surname><given-names>MS</given-names></name><name><surname>Peterson</surname><given-names>J</given-names></name><name><surname>Felsenfeld</surname><given-names>A</given-names></name><name><surname>Wetterstrand</surname><given-names>KA</given-names></name><name><surname>Patrinos</surname><given-names>A</given-names></name><name><surname>Morgan</surname><given-names>MJ</given-names></name><name><surname>de Jong</surname><given-names>P</given-names></name><name><surname>Catanese</surname><given-names>JJ</given-names></name><name><surname>Osoegawa</surname><given-names>K</given-names></name><name><surname>Shizuya</surname><given-names>H</given-names></name><name><surname>Choi</surname><given-names>S</given-names></name><name><surname>Chen</surname><given-names>YJ</given-names></name><name><surname>Szustakowki</surname><given-names>J</given-names></name><collab>International Human Genome Sequencing Consortium</collab></person-group><year iso-8601-date="2001">2001</year><article-title>Initial sequencing and analysis of the human genome</article-title><source>Nature</source><volume>409</volume><fpage>860</fpage><lpage>921</lpage><pub-id pub-id-type="doi">10.1038/35057062</pub-id><pub-id pub-id-type="pmid">11237011</pub-id></element-citation></ref><ref id="bib43"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Lazar</surname><given-names>T</given-names></name><name><surname>Tantos</surname><given-names>A</given-names></name><name><surname>Tompa</surname><given-names>P</given-names></name><name><surname>Schad</surname><given-names>E</given-names></name></person-group><year iso-8601-date="2022">2022</year><article-title>Intrinsic protein disorder uncouples affinity from binding specificity</article-title><source>Protein Science</source><volume>31</volume><elocation-id>e4455</elocation-id><pub-id pub-id-type="doi">10.1002/pro.4455</pub-id><pub-id pub-id-type="pmid">36305763</pub-id></element-citation></ref><ref id="bib44"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Li</surname><given-names>WH</given-names></name></person-group><year iso-8601-date="1987">1987</year><article-title>Models of nearly neutral mutations with particular implications for nonrandom usage of synonymous codons</article-title><source>Journal of Molecular Evolution</source><volume>24</volume><fpage>337</fpage><lpage>345</lpage><pub-id pub-id-type="doi">10.1007/BF02134132</pub-id><pub-id pub-id-type="pmid">3110426</pub-id></element-citation></ref><ref id="bib45"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname><given-names>SS</given-names></name><name><surname>Hockenberry</surname><given-names>AJ</given-names></name><name><surname>Jewett</surname><given-names>MC</given-names></name><name><surname>Amaral</surname><given-names>LAN</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>A novel framework for evaluating the performance of codon usage bias metrics</article-title><source>Journal of the Royal Society, Interface</source><volume>15</volume><elocation-id>20170667</elocation-id><pub-id pub-id-type="doi">10.1098/rsif.2017.0667</pub-id><pub-id pub-id-type="pmid">29386398</pub-id></element-citation></ref><ref id="bib46"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Long</surname><given-names>H</given-names></name><name><surname>Sung</surname><given-names>W</given-names></name><name><surname>Kucukyildirim</surname><given-names>S</given-names></name><name><surname>Williams</surname><given-names>E</given-names></name><name><surname>Miller</surname><given-names>SF</given-names></name><name><surname>Guo</surname><given-names>W</given-names></name><name><surname>Patterson</surname><given-names>C</given-names></name><name><surname>Gregory</surname><given-names>C</given-names></name><name><surname>Strauss</surname><given-names>C</given-names></name><name><surname>Stone</surname><given-names>C</given-names></name><name><surname>Berne</surname><given-names>C</given-names></name><name><surname>Kysela</surname><given-names>D</given-names></name><name><surname>Shoemaker</surname><given-names>WR</given-names></name><name><surname>Muscarella</surname><given-names>ME</given-names></name><name><surname>Luo</surname><given-names>H</given-names></name><name><surname>Lennon</surname><given-names>JT</given-names></name><name><surname>Brun</surname><given-names>YV</given-names></name><name><surname>Lynch</surname><given-names>M</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Evolutionary determinants of genome-wide nucleotide composition</article-title><source>Nature Ecology &amp; Evolution</source><volume>2</volume><fpage>237</fpage><lpage>240</lpage><pub-id pub-id-type="doi">10.1038/s41559-017-0425-y</pub-id></element-citation></ref><ref id="bib47"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Lynch</surname><given-names>M</given-names></name><name><surname>Ackerman</surname><given-names>MS</given-names></name><name><surname>Gout</surname><given-names>JF</given-names></name><name><surname>Long</surname><given-names>H</given-names></name><name><surname>Sung</surname><given-names>W</given-names></name><name><surname>Thomas</surname><given-names>WK</given-names></name><name><surname>Foster</surname><given-names>PL</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Genetic drift, selection and the evolution of the mutation rate</article-title><source>Nature Reviews. Genetics</source><volume>17</volume><fpage>704</fpage><lpage>714</lpage><pub-id pub-id-type="doi">10.1038/nrg.2016.104</pub-id><pub-id pub-id-type="pmid">27739533</pub-id></element-citation></ref><ref id="bib48"><element-citation publication-type="software"><person-group person-group-type="author"><collab>MaselLab</collab></person-group><year iso-8601-date="2024">2024</year><data-title>Codon-adaptation-index-of-species</data-title><version designator="swh:1:rev:408af3d150311c4732219abae67c6929421908df">swh:1:rev:408af3d150311c4732219abae67c6929421908df</version><source>Software Heritage</source><ext-link ext-link-type="uri" xlink:href="https://archive.softwareheritage.org/swh:1:dir:60841500b11a4aeb2f505d00dedae2a1db75f513;origin=https://github.com/MaselLab/Codon-Adaptation-Index-of-Species;visit=swh:1:snp:0ddc98c60b565558d38fd277d5e0a81e1e9fef0c;anchor=swh:1:rev:408af3d150311c4732219abae67c6929421908df">https://archive.softwareheritage.org/swh:1:dir:60841500b11a4aeb2f505d00dedae2a1db75f513;origin=https://github.com/MaselLab/Codon-Adaptation-Index-of-Species;visit=swh:1:snp:0ddc98c60b565558d38fd277d5e0a81e1e9fef0c;anchor=swh:1:rev:408af3d150311c4732219abae67c6929421908df</ext-link></element-citation></ref><ref id="bib49"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Meunier</surname><given-names>J</given-names></name><name><surname>Duret</surname><given-names>L</given-names></name></person-group><year iso-8601-date="2004">2004</year><article-title>Recombination drives the evolution of GC-content in the human genome</article-title><source>Molecular Biology and Evolution</source><volume>21</volume><fpage>984</fpage><lpage>990</lpage><pub-id pub-id-type="doi">10.1093/molbev/msh070</pub-id><pub-id pub-id-type="pmid">14963104</pub-id></element-citation></ref><ref id="bib50"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Novembre</surname><given-names>JA</given-names></name></person-group><year iso-8601-date="2002">2002</year><article-title>Accounting for background nucleotide composition when measuring codon usage bias</article-title><source>Molecular Biology and Evolution</source><volume>19</volume><fpage>1390</fpage><lpage>1394</lpage><pub-id pub-id-type="doi">10.1093/oxfordjournals.molbev.a004201</pub-id><pub-id pub-id-type="pmid">12140252</pub-id></element-citation></ref><ref id="bib51"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Novoa</surname><given-names>EM</given-names></name><name><surname>Jungreis</surname><given-names>I</given-names></name><name><surname>Jaillon</surname><given-names>O</given-names></name><name><surname>Kellis</surname><given-names>M</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Elucidation of codon usage signatures across the domains of life</article-title><source>Molecular Biology and Evolution</source><volume>36</volume><fpage>2328</fpage><lpage>2339</lpage><pub-id pub-id-type="doi">10.1093/molbev/msz124</pub-id><pub-id pub-id-type="pmid">31220870</pub-id></element-citation></ref><ref id="bib52"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ohta</surname><given-names>T</given-names></name></person-group><year iso-8601-date="1972">1972</year><article-title>Population size and rate of evolution</article-title><source>Journal of Molecular Evolution</source><volume>1</volume><fpage>305</fpage><lpage>314</lpage><pub-id pub-id-type="doi">10.1007/BF01653959</pub-id></element-citation></ref><ref id="bib53"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ohta</surname><given-names>T</given-names></name></person-group><year iso-8601-date="1973">1973</year><article-title>Slightly deleterious mutant substitutions in evolution</article-title><source>Nature</source><volume>246</volume><fpage>96</fpage><lpage>98</lpage><pub-id pub-id-type="doi">10.1038/246096a0</pub-id><pub-id pub-id-type="pmid">4585855</pub-id></element-citation></ref><ref id="bib54"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ohta</surname><given-names>T</given-names></name></person-group><year iso-8601-date="1992">1992</year><article-title>The nearly neutral theory of molecular evolution</article-title><source>Annual Review of Ecology and Systematics</source><volume>23</volume><fpage>263</fpage><lpage>286</lpage><pub-id pub-id-type="doi">10.1146/annurev.ecolsys.23.1.263</pub-id></element-citation></ref><ref id="bib55"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Paradis</surname><given-names>E</given-names></name><name><surname>Schliep</surname><given-names>K</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>ape 5.0: an environment for modern phylogenetics and evolutionary analyses in R</article-title><source>Bioinformatics</source><volume>35</volume><fpage>526</fpage><lpage>528</lpage><pub-id pub-id-type="doi">10.1093/bioinformatics/bty633</pub-id><pub-id pub-id-type="pmid">30016406</pub-id></element-citation></ref><ref id="bib56"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Plotkin</surname><given-names>JB</given-names></name><name><surname>Dushoff</surname><given-names>J</given-names></name><name><surname>Desai</surname><given-names>MM</given-names></name><name><surname>Fraser</surname><given-names>HB</given-names></name></person-group><year iso-8601-date="2006">2006</year><article-title>Codon usage and selection on proteins</article-title><source>Journal of Molecular Evolution</source><volume>63</volume><fpage>635</fpage><lpage>653</lpage><pub-id pub-id-type="doi">10.1007/s00239-005-0233-x</pub-id><pub-id pub-id-type="pmid">17043750</pub-id></element-citation></ref><ref id="bib57"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Plotkin</surname><given-names>JB</given-names></name><name><surname>Kudla</surname><given-names>G</given-names></name></person-group><year iso-8601-date="2011">2011</year><article-title>Synonymous but not the same: the causes and consequences of codon bias</article-title><source>Nature Reviews. Genetics</source><volume>12</volume><fpage>32</fpage><lpage>42</lpage><pub-id pub-id-type="doi">10.1038/nrg2899</pub-id><pub-id pub-id-type="pmid">21102527</pub-id></element-citation></ref><ref id="bib58"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Rocha</surname><given-names>EPC</given-names></name></person-group><year iso-8601-date="2004">2004</year><article-title>Codon usage bias from tRNA’s point of view: redundancy, specialization, and efficient decoding for translation optimization</article-title><source>Genome Research</source><volume>14</volume><fpage>2279</fpage><lpage>2286</lpage><pub-id pub-id-type="doi">10.1101/gr.2896904</pub-id><pub-id pub-id-type="pmid">15479947</pub-id></element-citation></ref><ref id="bib59"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Rohlf</surname><given-names>FJ</given-names></name></person-group><year iso-8601-date="2006">2006</year><article-title>A comment on phylogenetic correction</article-title><source>Evolution; International Journal of Organic Evolution</source><volume>60</volume><fpage>1509</fpage><lpage>1515</lpage><pub-id pub-id-type="doi">10.1554/05-550.1</pub-id><pub-id pub-id-type="pmid">16929667</pub-id></element-citation></ref><ref id="bib60"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Romiguier</surname><given-names>J</given-names></name><name><surname>Gayral</surname><given-names>P</given-names></name><name><surname>Ballenghien</surname><given-names>M</given-names></name><name><surname>Bernard</surname><given-names>A</given-names></name><name><surname>Cahais</surname><given-names>V</given-names></name><name><surname>Chenuil</surname><given-names>A</given-names></name><name><surname>Chiari</surname><given-names>Y</given-names></name><name><surname>Dernat</surname><given-names>R</given-names></name><name><surname>Duret</surname><given-names>L</given-names></name><name><surname>Faivre</surname><given-names>N</given-names></name><name><surname>Loire</surname><given-names>E</given-names></name><name><surname>Lourenco</surname><given-names>JM</given-names></name><name><surname>Nabholz</surname><given-names>B</given-names></name><name><surname>Roux</surname><given-names>C</given-names></name><name><surname>Tsagkogeorga</surname><given-names>G</given-names></name><name><surname>Weber</surname><given-names>AAT</given-names></name><name><surname>Weinert</surname><given-names>LA</given-names></name><name><surname>Belkhir</surname><given-names>K</given-names></name><name><surname>Bierne</surname><given-names>N</given-names></name><name><surname>Glémin</surname><given-names>S</given-names></name><name><surname>Galtier</surname><given-names>N</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>Comparative population genomics in animals uncovers the determinants of genetic diversity</article-title><source>Nature</source><volume>515</volume><fpage>261</fpage><lpage>263</lpage><pub-id pub-id-type="doi">10.1038/nature13685</pub-id><pub-id pub-id-type="pmid">25141177</pub-id></element-citation></ref><ref id="bib61"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Romiguier</surname><given-names>J</given-names></name><name><surname>Roux</surname><given-names>C</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>Analytical biases associated with GC-content in molecular evolution</article-title><source>Frontiers in Genetics</source><volume>8</volume><elocation-id>16</elocation-id><pub-id pub-id-type="doi">10.3389/fgene.2017.00016</pub-id><pub-id pub-id-type="pmid">28261263</pub-id></element-citation></ref><ref id="bib62"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Schad</surname><given-names>E</given-names></name><name><surname>Tompa</surname><given-names>P</given-names></name><name><surname>Hegyi</surname><given-names>H</given-names></name></person-group><year iso-8601-date="2011">2011</year><article-title>The relationship between proteome size, structural disorder and organism complexity</article-title><source>Genome Biology</source><volume>12</volume><elocation-id>R120</elocation-id><pub-id pub-id-type="doi">10.1186/gb-2011-12-12-r120</pub-id><pub-id pub-id-type="pmid">22182830</pub-id></element-citation></ref><ref id="bib63"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Sharp</surname><given-names>PM</given-names></name><name><surname>Li</surname><given-names>WH</given-names></name></person-group><year iso-8601-date="1987">1987</year><article-title>The codon Adaptation Index--a measure of directional synonymous codon usage bias, and its potential applications</article-title><source>Nucleic Acids Research</source><volume>15</volume><fpage>1281</fpage><lpage>1295</lpage><pub-id pub-id-type="doi">10.1093/nar/15.3.1281</pub-id><pub-id pub-id-type="pmid">3547335</pub-id></element-citation></ref><ref id="bib64"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Sharp</surname><given-names>PM</given-names></name><name><surname>Bailes</surname><given-names>E</given-names></name><name><surname>Grocock</surname><given-names>RJ</given-names></name><name><surname>Peden</surname><given-names>JF</given-names></name><name><surname>Sockett</surname><given-names>RE</given-names></name></person-group><year iso-8601-date="2005">2005</year><article-title>Variation in the strength of selected codon usage bias among bacteria</article-title><source>Nucleic Acids Research</source><volume>33</volume><fpage>1141</fpage><lpage>1153</lpage><pub-id pub-id-type="doi">10.1093/nar/gki242</pub-id><pub-id pub-id-type="pmid">15728743</pub-id></element-citation></ref><ref id="bib65"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Sharp</surname><given-names>PM</given-names></name><name><surname>Emery</surname><given-names>LR</given-names></name><name><surname>Zeng</surname><given-names>K</given-names></name></person-group><year iso-8601-date="2010">2010</year><article-title>Forces that influence the evolution of codon bias</article-title><source>Philosophical Transactions of the Royal Society B</source><volume>365</volume><fpage>1203</fpage><lpage>1212</lpage><pub-id pub-id-type="doi">10.1098/rstb.2009.0305</pub-id></element-citation></ref><ref id="bib66"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Smith</surname><given-names>NGC</given-names></name><name><surname>Eyre-Walker</surname><given-names>A</given-names></name></person-group><year iso-8601-date="2001">2001</year><article-title>Synonymous codon bias is not caused by mutation bias in G+C-rich genes in humans</article-title><source>Molecular Biology and Evolution</source><volume>18</volume><fpage>982</fpage><lpage>986</lpage><pub-id pub-id-type="doi">10.1093/oxfordjournals.molbev.a003899</pub-id></element-citation></ref><ref id="bib67"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Subramanian</surname><given-names>S</given-names></name></person-group><year iso-8601-date="2008">2008</year><article-title>Nearly neutrality and the evolution of codon usage bias in eukaryotic genomes</article-title><source>Genetics</source><volume>178</volume><fpage>2429</fpage><lpage>2432</lpage><pub-id pub-id-type="doi">10.1534/genetics.107.086405</pub-id><pub-id pub-id-type="pmid">18430960</pub-id></element-citation></ref><ref id="bib68"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Sun</surname><given-names>X</given-names></name><name><surname>Yang</surname><given-names>Q</given-names></name><name><surname>Xia</surname><given-names>X</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>An improved implementation of effective number of codons (nc)</article-title><source>Molecular Biology and Evolution</source><volume>30</volume><fpage>191</fpage><lpage>196</lpage><pub-id pub-id-type="doi">10.1093/molbev/mss201</pub-id><pub-id pub-id-type="pmid">22915832</pub-id></element-citation></ref><ref id="bib69"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Theillet</surname><given-names>FX</given-names></name><name><surname>Kalmar</surname><given-names>L</given-names></name><name><surname>Tompa</surname><given-names>P</given-names></name><name><surname>Han</surname><given-names>KH</given-names></name><name><surname>Selenko</surname><given-names>P</given-names></name><name><surname>Dunker</surname><given-names>AK</given-names></name><name><surname>Daughdrill</surname><given-names>GW</given-names></name><name><surname>Uversky</surname><given-names>VN</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>The alphabet of intrinsic disorder</article-title><source>Intrinsically Disordered Proteins</source><volume>1</volume><elocation-id>e24360</elocation-id><pub-id pub-id-type="doi">10.4161/idp.24360</pub-id></element-citation></ref><ref id="bib70"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Urrutia</surname><given-names>AO</given-names></name><name><surname>Hurst</surname><given-names>LD</given-names></name></person-group><year iso-8601-date="2001">2001</year><article-title>Codon usage bias covaries with expression breadth and the rate of synonymous evolution in humans, but this is not evidence for selection</article-title><source>Genetics</source><volume>159</volume><fpage>1191</fpage><lpage>1199</lpage><pub-id pub-id-type="doi">10.1093/genetics/159.3.1191</pub-id><pub-id pub-id-type="pmid">11729162</pub-id></element-citation></ref><ref id="bib71"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Vavouri</surname><given-names>T</given-names></name><name><surname>Semple</surname><given-names>JI</given-names></name><name><surname>Garcia-Verdugo</surname><given-names>R</given-names></name><name><surname>Lehner</surname><given-names>B</given-names></name></person-group><year iso-8601-date="2009">2009</year><article-title>Intrinsic protein disorder and interaction promiscuity are widely associated with dosage sensitivity</article-title><source>Cell</source><volume>138</volume><fpage>198</fpage><lpage>208</lpage><pub-id pub-id-type="doi">10.1016/j.cell.2009.04.029</pub-id><pub-id pub-id-type="pmid">19596244</pub-id></element-citation></ref><ref id="bib72"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Vicario</surname><given-names>S</given-names></name><name><surname>Moriyama</surname><given-names>EN</given-names></name><name><surname>Powell</surname><given-names>JR</given-names></name></person-group><year iso-8601-date="2007">2007</year><article-title>Codon usage in twelve species of <italic>Drosophila</italic></article-title><source>BMC Evolutionary Biology</source><volume>7</volume><elocation-id>226</elocation-id><pub-id pub-id-type="doi">10.1186/1471-2148-7-226</pub-id><pub-id pub-id-type="pmid">18005411</pub-id></element-citation></ref><ref id="bib73"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Wangen</surname><given-names>JR</given-names></name><name><surname>Green</surname><given-names>R</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>Stop codon context influences genome-wide stimulation of termination codon readthrough by aminoglycosides</article-title><source>eLife</source><volume>9</volume><elocation-id>e52611</elocation-id><pub-id pub-id-type="doi">10.7554/eLife.52611</pub-id></element-citation></ref><ref id="bib74"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Wright</surname><given-names>F</given-names></name></person-group><year iso-8601-date="1990">1990</year><article-title>The “effective number of codons” used in a gene</article-title><source>Gene</source><volume>87</volume><fpage>23</fpage><lpage>29</lpage><pub-id pub-id-type="doi">10.1016/0378-1119(90)90491-9</pub-id><pub-id pub-id-type="pmid">2110097</pub-id></element-citation></ref><ref id="bib75"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Xue</surname><given-names>B</given-names></name><name><surname>Dunker</surname><given-names>AK</given-names></name><name><surname>Uversky</surname><given-names>VN</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>Orderly order in protein intrinsic disorder distribution: disorder in 3500 proteomes from viruses and the three domains of life</article-title><source>Journal of Biomolecular Structure &amp; Dynamics</source><volume>30</volume><fpage>137</fpage><lpage>149</lpage><pub-id pub-id-type="doi">10.1080/07391102.2012.675145</pub-id><pub-id pub-id-type="pmid">22702725</pub-id></element-citation></ref><ref id="bib76"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname><given-names>Z</given-names></name><name><surname>Li</surname><given-names>J</given-names></name><name><surname>Cui</surname><given-names>P</given-names></name><name><surname>Ding</surname><given-names>F</given-names></name><name><surname>Li</surname><given-names>A</given-names></name><name><surname>Townsend</surname><given-names>JP</given-names></name><name><surname>Yu</surname><given-names>J</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>Codon deviation coefficient: a novel measure for estimating codon usage bias and its statistical significance</article-title><source>BMC Bioinformatics</source><volume>13</volume><elocation-id>43</elocation-id><pub-id pub-id-type="doi">10.1186/1471-2105-13-43</pub-id><pub-id pub-id-type="pmid">22435713</pub-id></element-citation></ref></ref-list></back><sub-article article-type="editor-report" id="sa0"><front-stub><article-id pub-id-type="doi">10.7554/eLife.87335.3.sa0</article-id><title-group><article-title>eLife assessment</article-title></title-group><contrib-group><contrib contrib-type="author"><name><surname>Ogbunugafor</surname><given-names>C Brandon</given-names></name><role specific-use="editor">Reviewing Editor</role><aff><institution>Yale University</institution><country>United States</country></aff></contrib></contrib-group><kwd-group kwd-group-type="evidence-strength"><kwd>Solid</kwd></kwd-group><kwd-group kwd-group-type="claim-importance"><kwd>Useful</kwd></kwd-group></front-stub><body><p>This study develops a <bold>useful</bold> metric for quantifying codon usage adaptation - the Codon Adaptation Index of Species (CAIS). This metric permits direct comparisons of the strength of selection at the molecular level across species. The study is based on <bold>solid</bold> evidence, and the authors identify relationships between CAIS and the presence of disordered protein domains. Other correlations, such as the one between CAIS and body size, are weak and non-significant. In summary, the study introduces an interesting new approach to quantifying codon usage across species, which may be helpful in attempts to measure selection at the molecular level.</p></body></sub-article><sub-article article-type="referee-report" id="sa1"><front-stub><article-id pub-id-type="doi">10.7554/eLife.87335.3.sa1</article-id><title-group><article-title>Reviewer #2 (Public Review):</article-title></title-group><contrib-group><contrib contrib-type="author"><anonymous/><role specific-use="referee">Reviewer</role></contrib></contrib-group></front-stub><body><p>Summary:</p><p>The goal of the authors in this study is to develop a more reliable approach for quantifying codon usage such that it is more comparable across species. Specifically, the authors wish to estimate the degree of adaptive codon usage, which is potentially a general proxy for the strength of selection at the molecular level. To this end, the authors created the Codon Adaptation Index for Species (CAIS) that attempts to control for differences in amino acid usage and GC% across species. Using their new metric, the authors observe a positive relationship between CAIS and the overall “disorderedness” of a species protein domains. I think CAIS has the potential to be a valuable tool for those interested in comparing codon adaptation across species in certain situations. However, I have certain theoretical concerns about CAIS as a direct proxy for the efficiency of selection sNe when mutation bias changes across species.</p><p>Strengths:</p><p>(1) I appreciate that the authors recognize the potential issues of comparing CAI when amino acid usage varies and correct for this in CAIS. I think this is sometimes an under-appreciated point in the codon usage literature, as CAI is a relative measure of codon usage bias (i.e. only considers synonyms). However, the strength of natural selection on codon usage can potentially vary across amino acids, such that comparing mean CAI between protein regions with different amino acid biases may result in spurious signals of statistical significance.</p><p>(2) The CAIS metric presented here is generally applicable to any species that has an annotated genome with protein-coding sequences. A significant improvement over the previous version is the implementation of software tool for applying this method.</p><p>(3) The authors do a better job of putting their results in the context of the underlying theory of CAIS compared to the previous version.</p><p>(4) The paper is generally well-written.</p><p>Weaknesses:</p><p>(1) The previously observed correlation between CAIS and body size was due to a bug when calculating phylogenetic independent contrasts. I commend the authors for acknowledging this mistake and updating the manuscript accordingly. I feel that the unobserved correlation between CAIS and body size should remain in the final version of the manuscript. Although it is disappointing that it is not statistically significant, the corrected results are consistent with previous findings (Kessler and Dean 2014).</p><p>(2) I appreciate the authors for providing a more detailed explanation of the theoretical basis model. However, I remain skeptical that shifts in CAIS across species indicates shifts in the strength of selection. I am leaving the math from my previous review here for completeness.</p><p>As in my previous review, let’s take a closer look at the ratio of observed codon frequencies vs. expected codon frequencies under mutation alone, which was previously notated as RSCUS in the original formulation. In this review, I will keep using the RSCUS notation, even though it has been dropped from the updated version. The key point is this is the ratio of observed and expected codon frequencies. If this ratio is 1 for all codons, then CAIS would be 0 based on equation 7 in the manuscript – consistent with the complete absence of selection on codon usage. From here on out, subscripts will only be used to denote the codon and it will be assumed that we are only considering the case of r = genome for some species s.<disp-formula id="sa1equ1"><mml:math id="sa1m1"><mml:mrow><mml:mi>R</mml:mi><mml:mi>S</mml:mi><mml:mi>C</mml:mi><mml:mi>U</mml:mi><mml:msub><mml:mi>S</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mfrac><mml:msub><mml:mi>O</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mi>E</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mfrac></mml:mrow></mml:math></disp-formula></p><p>I think what the authors are attempting to do is “divide out” the effects of mutation bias (as given by Ei), such that only the effects of natural selection remain, i.e. deviations from the expected frequency based on mutation bias alone represents adaptive codon usage. Consider Gilchrist et al. GBE 2015, which says that the expected frequency of codon i at selection-mutation-drift equilibrium in gene g for an amino acid with Na synonymous codons is<disp-formula id="sa1equ2"><mml:math id="sa1m2"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:msub><mml:mi>E</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>g</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mfrac><mml:msup><mml:mi>e</mml:mi><mml:mrow><mml:mo>−</mml:mo><mml:mi mathvariant="normal">Δ</mml:mi><mml:msub><mml:mi>M</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>−</mml:mo><mml:mi mathvariant="normal">Δ</mml:mi><mml:msub><mml:mi>η</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mi>ϕ</mml:mi><mml:mrow><mml:mi>g</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msup><mml:mrow><mml:munderover><mml:mo>∑</mml:mo><mml:mrow><mml:mi>j</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:msub><mml:mi>N</mml:mi><mml:mrow><mml:mi>a</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:munderover><mml:msup><mml:mi>e</mml:mi><mml:mrow><mml:mo>−</mml:mo><mml:mi mathvariant="normal">Δ</mml:mi><mml:msub><mml:mi>M</mml:mi><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>−</mml:mo><mml:mi mathvariant="normal">Δ</mml:mi><mml:msub><mml:mi>η</mml:mi><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mi>ϕ</mml:mi><mml:mrow><mml:mi>g</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msup></mml:mrow></mml:mfrac></mml:mstyle></mml:mrow></mml:math></disp-formula></p><p>where ∆M is the mutation bias, ∆η is the strength of selection scaled by the strength of drift, and φg is the gene expression level of gene g. In this case, ∆M and ∆η reflect the strength and direction of mutation bias and natural selection relative to a reference codon, for which ∆M,∆η = 0. Assuming the selection-mutation-drift equilibrium model is generally adequate to model of the true codon usage patterns in a genome (as I do and I think the authors do, too), the Ei,g could be considered the expected observed frequency codon i in gene g</p><p>E[Oi,g].</p><p>Let’s re-write the <inline-formula><mml:math id="sa1m3"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:msub><mml:mi>E</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mfrac><mml:msub><mml:mi>p</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:munderover><mml:mo>∑</mml:mo><mml:mrow><mml:mi>j</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:msub><mml:mi>N</mml:mi><mml:mrow><mml:mi>a</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:munderover><mml:msub><mml:mi>p</mml:mi><mml:mrow><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfrac></mml:mrow></mml:mstyle></mml:math></inline-formula> in the form of Gilchrist et al., such that it is a function of mutation bias ∆M. For simplicity we will consider just the two codon case and assume the amino acid sequence is fixed. Assuming GC% is at equilibrium, the term gr and 1 − gr can be written as<disp-formula id="sa1equ3"><mml:math id="sa1m4"><mml:mrow><mml:mtable columnalign="right left right left right left right left right left right left" rowspacing="3pt" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true"><mml:mtr><mml:mtd><mml:msub><mml:mi>g</mml:mi><mml:mrow><mml:mi>r</mml:mi></mml:mrow></mml:msub></mml:mtd><mml:mtd><mml:mi/><mml:mo>=</mml:mo><mml:mfrac><mml:msub><mml:mi>μ</mml:mi><mml:mrow><mml:mi>A</mml:mi><mml:mi>T</mml:mi><mml:mo stretchy="false">→</mml:mo><mml:mi>G</mml:mi><mml:mi>C</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:msub><mml:mi>μ</mml:mi><mml:mrow><mml:mi>A</mml:mi><mml:mi>T</mml:mi><mml:mo stretchy="false">→</mml:mo><mml:mi>G</mml:mi><mml:mi>C</mml:mi></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mi>μ</mml:mi><mml:mrow><mml:mi>G</mml:mi><mml:mi>C</mml:mi><mml:mo stretchy="false">→</mml:mo><mml:mi>A</mml:mi><mml:mi>T</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfrac></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mn>1</mml:mn><mml:mo>−</mml:mo><mml:msub><mml:mi>g</mml:mi><mml:mrow><mml:mi>r</mml:mi></mml:mrow></mml:msub></mml:mtd><mml:mtd><mml:mi/><mml:mo>=</mml:mo><mml:mfrac><mml:msub><mml:mi>μ</mml:mi><mml:mrow><mml:mi>G</mml:mi><mml:mi>C</mml:mi><mml:mo stretchy="false">→</mml:mo><mml:mi>A</mml:mi><mml:mi>T</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:msub><mml:mi>μ</mml:mi><mml:mrow><mml:mi>A</mml:mi><mml:mi>T</mml:mi><mml:mo stretchy="false">→</mml:mo><mml:mi>G</mml:mi><mml:mi>C</mml:mi></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mi>μ</mml:mi><mml:mrow><mml:mi>G</mml:mi><mml:mi>C</mml:mi><mml:mo stretchy="false">→</mml:mo><mml:mi>A</mml:mi><mml:mi>T</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfrac></mml:mtd></mml:mtr></mml:mtable></mml:mrow></mml:math></disp-formula></p><p>where µx→y is the mutation rate from nucleotides x to y. As described in Gilchrist et al. MBE 2015 and</p><p>Shah and Gilchrist PNAS 2011, the mutation bias <inline-formula><mml:math id="sa1m5"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi mathvariant="normal">Δ</mml:mi><mml:msub><mml:mi>M</mml:mi><mml:mrow><mml:mi>N</mml:mi><mml:mi>N</mml:mi><mml:mi>A</mml:mi><mml:mo>,</mml:mo><mml:mi>N</mml:mi><mml:mi>N</mml:mi><mml:mi>G</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mi>log</mml:mi><mml:mo>⁡</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mfrac><mml:msub><mml:mi>μ</mml:mi><mml:mrow><mml:mi>A</mml:mi><mml:mi>T</mml:mi><mml:mo stretchy="false">→</mml:mo><mml:mi>G</mml:mi><mml:mi>C</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mi>μ</mml:mi><mml:mrow><mml:mi>G</mml:mi><mml:mi>C</mml:mi><mml:mo stretchy="false">→</mml:mo><mml:mi>A</mml:mi><mml:mi>T</mml:mi></mml:mrow></mml:msub></mml:mfrac><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:mstyle></mml:math></inline-formula> .This can be expressed in terms of the equilibrium GC content by recognizing that<disp-formula id="sa1equ4"><mml:math id="sa1m6"><mml:mrow><mml:mtable columnalign="right left right left right left right left right left right left" rowspacing="3pt" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true"><mml:mtr><mml:mtd><mml:mfrac><mml:msub><mml:mi>g</mml:mi><mml:mrow><mml:mi>r</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mn>1</mml:mn><mml:mo>−</mml:mo><mml:msub><mml:mi>g</mml:mi><mml:mrow><mml:mi>r</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfrac></mml:mtd><mml:mtd><mml:mi/><mml:mo>=</mml:mo><mml:mfrac><mml:msub><mml:mi>μ</mml:mi><mml:mrow><mml:mi>A</mml:mi><mml:mi>T</mml:mi><mml:mo stretchy="false">→</mml:mo><mml:mi>G</mml:mi><mml:mi>C</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mi>μ</mml:mi><mml:mrow><mml:mi>G</mml:mi><mml:mi>C</mml:mi><mml:mo stretchy="false">→</mml:mo><mml:mi>A</mml:mi><mml:mi>T</mml:mi></mml:mrow></mml:msub></mml:mfrac></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mo stretchy="false">⟹</mml:mo><mml:mfrac><mml:msub><mml:mi>g</mml:mi><mml:mrow><mml:mi>r</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mn>1</mml:mn><mml:mo>−</mml:mo><mml:msub><mml:mi>g</mml:mi><mml:mrow><mml:mi>r</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfrac></mml:mtd><mml:mtd><mml:mi/><mml:mo>=</mml:mo><mml:msup><mml:mi>e</mml:mi><mml:mrow><mml:mi mathvariant="normal">Δ</mml:mi><mml:mi>M</mml:mi></mml:mrow></mml:msup></mml:mtd></mml:mtr></mml:mtable></mml:mrow></mml:math></disp-formula></p><p>As we are assuming the amino acid sequence is fixed, the probability of observing a synonymous codon i at an amino acid becomes just a Bernoulli process.<disp-formula id="sa1equ5"><mml:math id="sa1m7"><mml:mrow><mml:msub><mml:mi>p</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:msubsup><mml:mi>g</mml:mi><mml:mrow><mml:mi>r</mml:mi></mml:mrow><mml:mrow><mml:mi>x</mml:mi></mml:mrow></mml:msubsup><mml:msup><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>−</mml:mo><mml:msub><mml:mi>g</mml:mi><mml:mrow><mml:mi>r</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mn>1</mml:mn><mml:mo>−</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:msup></mml:mrow></mml:math></disp-formula></p><p>If we do this, then<disp-formula id="sa1equ6"><mml:math id="sa1m8"><mml:mrow><mml:mtable columnalign="right left right left right left right left right left right left" rowspacing="3pt" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true"><mml:mtr><mml:mtd><mml:msub><mml:mi>E</mml:mi><mml:mrow><mml:mi>N</mml:mi><mml:mi>N</mml:mi><mml:mi>A</mml:mi></mml:mrow></mml:msub></mml:mtd><mml:mtd><mml:mi/><mml:mo>=</mml:mo><mml:mfrac><mml:msub><mml:mi>p</mml:mi><mml:mrow><mml:mi>N</mml:mi><mml:mi>N</mml:mi><mml:mi>A</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:msub><mml:mi>p</mml:mi><mml:mrow><mml:mi>N</mml:mi><mml:mi>N</mml:mi><mml:mi>A</mml:mi></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mi>p</mml:mi><mml:mrow><mml:mi>N</mml:mi><mml:mi>N</mml:mi><mml:mi>G</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfrac></mml:mtd></mml:mtr><mml:mtr><mml:mtd/><mml:mtd><mml:mi/><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn><mml:mo>−</mml:mo><mml:msub><mml:mi>g</mml:mi><mml:mrow><mml:mi>r</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:msub><mml:mi>g</mml:mi><mml:mrow><mml:mi>r</mml:mi></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mn>1</mml:mn><mml:mo>−</mml:mo><mml:msub><mml:mi>g</mml:mi><mml:mrow><mml:mi>r</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:mfrac></mml:mtd></mml:mtr><mml:mtr><mml:mtd/><mml:mtd><mml:mi/><mml:mo>=</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mrow><mml:mfrac><mml:msub><mml:mi>g</mml:mi><mml:mrow><mml:mi>r</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:mn>1</mml:mn><mml:mo>−</mml:mo><mml:msub><mml:mi>g</mml:mi><mml:mrow><mml:mi>r</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfrac><mml:mo>+</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:mfrac></mml:mtd></mml:mtr><mml:mtr><mml:mtd/><mml:mtd><mml:mi/><mml:mo>=</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mrow><mml:msup><mml:mi>e</mml:mi><mml:mrow><mml:mi mathvariant="normal">Δ</mml:mi><mml:mi>M</mml:mi></mml:mrow></mml:msup><mml:mo>+</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:mfrac></mml:mtd></mml:mtr><mml:mtr><mml:mtd/><mml:mtd><mml:mi/><mml:mo>=</mml:mo><mml:mfrac><mml:msup><mml:mi>e</mml:mi><mml:mrow><mml:mo>−</mml:mo><mml:mi mathvariant="normal">Δ</mml:mi><mml:mi>M</mml:mi></mml:mrow></mml:msup><mml:mrow><mml:mn>1</mml:mn><mml:mo>+</mml:mo><mml:msup><mml:mi>e</mml:mi><mml:mrow><mml:mo>−</mml:mo><mml:mi mathvariant="normal">Δ</mml:mi><mml:mi>M</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:mfrac></mml:mtd></mml:mtr></mml:mtable></mml:mrow></mml:math></disp-formula></p><p>Recall that in the Gilchrist et al. framework, the reference codon has ∆MNNG,NNG = 0 = ⇒ e−∆MNNG,NNG =</p><p>(1) Thus, we have recovered the Gilchrist et al. model from the formulation of Ei under the assumption that natural selection has no impact on codon usage and codon NNG is the pre-defined reference codon. To see this, plug in 0 for ∆η in equation (1).</p><p>We can then calculate the expected RSCUS using equation (1) (using notation E[Oi]) and equation (6) for the two codon case. For simplicity assume, we are only considering a gene of average expression defined as <inline-formula><mml:math id="sa1m9"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msub><mml:mi>ϕ</mml:mi><mml:mrow><mml:mi>g</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:mstyle></mml:math></inline-formula>. Assume in this case that NNG is the reference codon (∆MNNG,∆ηNNG = 0).<disp-formula id="sa1equ7"><mml:math id="sa1m10"><mml:mrow><mml:mtable columnalign="right left right left right left right left right left right left" rowspacing="3pt" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true"><mml:mtr><mml:mtd><mml:mi>E</mml:mi><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mi>R</mml:mi><mml:mi>S</mml:mi><mml:mi>C</mml:mi><mml:mi>U</mml:mi><mml:msub><mml:mi>S</mml:mi><mml:mrow><mml:mi>N</mml:mi><mml:mi>N</mml:mi><mml:mi>A</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:mtd><mml:mtd><mml:mi/><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>E</mml:mi><mml:mrow><mml:mo>[</mml:mo><mml:msub><mml:mi>O</mml:mi><mml:mrow><mml:mi>N</mml:mi><mml:mi>N</mml:mi><mml:mi>A</mml:mi></mml:mrow></mml:msub><mml:mo>]</mml:mo></mml:mrow></mml:mrow><mml:msub><mml:mi>E</mml:mi><mml:mrow><mml:mi>N</mml:mi><mml:mi>N</mml:mi><mml:mi>A</mml:mi></mml:mrow></mml:msub></mml:mfrac></mml:mtd></mml:mtr><mml:mtr><mml:mtd/><mml:mtd><mml:mi/><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:msup><mml:mi>e</mml:mi><mml:mrow><mml:mo>−</mml:mo><mml:mi mathvariant="normal">Δ</mml:mi><mml:msub><mml:mi>η</mml:mi><mml:mrow><mml:mi>N</mml:mi><mml:mi>N</mml:mi><mml:mi>A</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msup><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:msup><mml:mi>e</mml:mi><mml:mrow><mml:mo>−</mml:mo><mml:mi mathvariant="normal">Δ</mml:mi><mml:msub><mml:mi>M</mml:mi><mml:mrow><mml:mi>N</mml:mi><mml:mi>N</mml:mi><mml:mi>A</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msup><mml:mo>+</mml:mo><mml:msup><mml:mi>e</mml:mi><mml:mrow><mml:mo>−</mml:mo><mml:mi mathvariant="normal">Δ</mml:mi><mml:msub><mml:mi>M</mml:mi><mml:mrow><mml:mi>N</mml:mi><mml:mi>N</mml:mi><mml:mi>G</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msup></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:msup><mml:mi>e</mml:mi><mml:mrow><mml:mo>−</mml:mo><mml:mi mathvariant="normal">Δ</mml:mi><mml:msub><mml:mi>M</mml:mi><mml:mrow><mml:mi>N</mml:mi><mml:mi>N</mml:mi><mml:mi>A</mml:mi></mml:mrow></mml:msub><mml:mo>−</mml:mo><mml:mi mathvariant="normal">Δ</mml:mi><mml:msub><mml:mi>η</mml:mi><mml:mrow><mml:mi>N</mml:mi><mml:mi>N</mml:mi><mml:mi>A</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msup><mml:mo>+</mml:mo><mml:msup><mml:mi>e</mml:mi><mml:mrow><mml:mo>−</mml:mo><mml:mi mathvariant="normal">Δ</mml:mi><mml:msub><mml:mi>M</mml:mi><mml:mrow><mml:mi>N</mml:mi><mml:mi>N</mml:mi><mml:mi>G</mml:mi></mml:mrow></mml:msub><mml:mo>−</mml:mo><mml:mi mathvariant="normal">Δ</mml:mi><mml:msub><mml:mi>η</mml:mi><mml:mrow><mml:mi>N</mml:mi><mml:mi>N</mml:mi><mml:mi>G</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msup></mml:mrow></mml:mfrac></mml:mtd></mml:mtr><mml:mtr><mml:mtd/><mml:mtd><mml:mi/><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:msup><mml:mi>e</mml:mi><mml:mrow><mml:mo>−</mml:mo><mml:mi mathvariant="normal">Δ</mml:mi><mml:msub><mml:mi>M</mml:mi><mml:mrow><mml:mi>N</mml:mi><mml:mi>N</mml:mi><mml:mi>A</mml:mi></mml:mrow></mml:msub><mml:mo>−</mml:mo><mml:mi mathvariant="normal">Δ</mml:mi><mml:msub><mml:mi>η</mml:mi><mml:mrow><mml:mi>N</mml:mi><mml:mi>N</mml:mi><mml:mi>A</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msup><mml:mo>+</mml:mo><mml:msup><mml:mi>e</mml:mi><mml:mrow><mml:mo>−</mml:mo><mml:mi mathvariant="normal">Δ</mml:mi><mml:msub><mml:mi>M</mml:mi><mml:mrow><mml:mi>N</mml:mi><mml:mi>N</mml:mi><mml:mi>G</mml:mi></mml:mrow></mml:msub><mml:mo>−</mml:mo><mml:mi mathvariant="normal">Δ</mml:mi><mml:msub><mml:mi>η</mml:mi><mml:mrow><mml:mi>N</mml:mi><mml:mi>N</mml:mi><mml:mi>A</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msup></mml:mrow><mml:mrow><mml:msup><mml:mi>e</mml:mi><mml:mrow><mml:mo>−</mml:mo><mml:mi mathvariant="normal">Δ</mml:mi><mml:msub><mml:mi>M</mml:mi><mml:mrow><mml:mi>N</mml:mi><mml:mi>N</mml:mi><mml:mi>A</mml:mi></mml:mrow></mml:msub><mml:mo>−</mml:mo><mml:mi mathvariant="normal">Δ</mml:mi><mml:msub><mml:mi>η</mml:mi><mml:mrow><mml:mi>N</mml:mi><mml:mi>N</mml:mi><mml:mi>A</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msup><mml:mo>+</mml:mo><mml:msup><mml:mi>e</mml:mi><mml:mrow><mml:mo>−</mml:mo><mml:mi mathvariant="normal">Δ</mml:mi><mml:msub><mml:mi>M</mml:mi><mml:mrow><mml:mi>N</mml:mi><mml:mi>N</mml:mi><mml:mi>G</mml:mi></mml:mrow></mml:msub><mml:mo>−</mml:mo><mml:mi mathvariant="normal">Δ</mml:mi><mml:msub><mml:mi>η</mml:mi><mml:mrow><mml:mi>N</mml:mi><mml:mi>N</mml:mi><mml:mi>G</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msup></mml:mrow></mml:mfrac></mml:mtd></mml:mtr><mml:mtr><mml:mtd/><mml:mtd><mml:mi/><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:msup><mml:mi>e</mml:mi><mml:mrow><mml:mo>−</mml:mo><mml:mi mathvariant="normal">Δ</mml:mi><mml:msub><mml:mi>M</mml:mi><mml:mrow><mml:mi>N</mml:mi><mml:mi>N</mml:mi><mml:mi>A</mml:mi></mml:mrow></mml:msub><mml:mo>−</mml:mo><mml:mi mathvariant="normal">Δ</mml:mi><mml:msub><mml:mi>η</mml:mi><mml:mrow><mml:mi>N</mml:mi><mml:mi>N</mml:mi><mml:mi>A</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msup><mml:mo>+</mml:mo><mml:msup><mml:mi>e</mml:mi><mml:mrow><mml:mo>−</mml:mo><mml:mi mathvariant="normal">Δ</mml:mi><mml:msub><mml:mi>η</mml:mi><mml:mrow><mml:mi>N</mml:mi><mml:mi>N</mml:mi><mml:mi>A</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msup></mml:mrow><mml:mrow><mml:msup><mml:mi>e</mml:mi><mml:mrow><mml:mo>−</mml:mo><mml:mi mathvariant="normal">Δ</mml:mi><mml:msub><mml:mi>M</mml:mi><mml:mrow><mml:mi>N</mml:mi><mml:mi>N</mml:mi><mml:mi>A</mml:mi></mml:mrow></mml:msub><mml:mo>−</mml:mo><mml:mi mathvariant="normal">Δ</mml:mi><mml:msub><mml:mi>η</mml:mi><mml:mrow><mml:mi>N</mml:mi><mml:mi>N</mml:mi><mml:mi>A</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msup><mml:mo>+</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:mfrac></mml:mtd></mml:mtr></mml:mtable></mml:mrow></mml:math></disp-formula></p><p>This shows that the expected value of RSCUS for a two codon amino acid is expected to increase as the strength of selection ∆η increases, which is desired. Note that ∆η in Gilchrist et al. is formulated in terms of selection against a codon relative to the reference, such that a negative value represents that a codon is favored relative to the reference. If ∆η = 0 (i.e. selection does not favor either codon), then E[RSCUS] = 1. Also note that the expected RSCUS does not remain independent of the mutation bias. This means that even if sNe (i.e. the strength of natural selection) does not change between species, changes to the strength and direction of mutation bias across species could impact RSCUS. Assuming my math is right, I think one needs to be cautious when interpreting CAIS as representative of the differences in the efficiency of selection across species except under very particular circumstances.</p><p>Consider our 2-codon amino acid scenario. You can see how changing GC content without changing selection can alter the CAIS values calculated from these two codons. Particularly problematic appears to be cases of extreme mutation biases, where CAIS tends toward 0 even for higher absolute values of the selection parameter. Codon usage for the majority of the genome will be primarily determined by mutation biases,</p><p>with selection being generally strongest in a relatively few highly-expressed genes. Strong enough mutation biases ultimately can overwhelm selection, even in highly-expressed genes, reducing the fraction of sites subject to codon adaptation.</p><fig position="float" id="sa1fig1"><label>Review image 1.</label><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-87335-sa1-fig1-v1.tif"/></fig><fig position="float" id="sa1fig2"><label>Review image 2.</label><caption><title>CAIS (Low Expression).</title></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-87335-sa1-fig2-v1.tif"/></fig><fig position="float" id="sa1fig3"><label>Review image 3.</label><caption><title>CAIS (Average Expression).</title></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-87335-sa1-fig3-v1.tif"/></fig><fig position="float" id="sa1fig4"><label>Review image 4.</label><caption><title>CAIS (High Expression).</title></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-87335-sa1-fig4-v1.tif"/></fig><p>If we treat the expected codon frequencies as genome-wide frequencies, then we are basically assuming this genome made up entirely of a single 2-codon amino acid with selection on codon usage being uniform across all genes. This is obviously not true, but I think it shows some of the potential limitations of the CAIS approach. Based on these simulations, CAIS seems best employed under specific scenarios. One such case could be when it is known that mutation bias varies little across the species of interest. Looking at the species used in this manuscript, most of them have a GC content around 0.41, so I suspect their results are okay (assuming things like GC-biased gene conversion are not an issue). Outliers in GC content probably are best excluded from the analysis.</p><p>Although I have not done so, I am sure this could be extended to the 4 and 6 codon amino acids. One potential challenge to CAIS is the non-monotonic changes in codon frequencies observed in some species (again, see Shah and Gilchrist 2011 and Gilchrist et al. 2015).</p></body></sub-article><sub-article article-type="author-comment" id="sa2"><front-stub><article-id pub-id-type="doi">10.7554/eLife.87335.3.sa2</article-id><title-group><article-title>Author response</article-title></title-group><contrib-group><contrib contrib-type="author"><name><surname>Weibel</surname><given-names>Catherine A</given-names></name><role specific-use="author">Author</role><aff><institution>Stanford University</institution><addr-line><named-content content-type="city">Tucson</named-content></addr-line><country>United States</country></aff></contrib><contrib contrib-type="author"><name><surname>Wheeler</surname><given-names>Andrew L</given-names></name><role specific-use="author">Author</role><aff><institution>University of Arizona</institution><addr-line><named-content content-type="city">Tucson</named-content></addr-line><country>United States</country></aff></contrib><contrib contrib-type="author"><name><surname>James</surname><given-names>Jennifer E</given-names></name><role specific-use="author">Author</role><aff><institution>University of Arizona</institution><addr-line><named-content content-type="city">Tucson</named-content></addr-line><country>United States</country></aff></contrib><contrib contrib-type="author"><name><surname>Willis</surname><given-names>Sara M</given-names></name><role specific-use="author">Author</role><aff><institution>University of Arizona</institution><addr-line><named-content content-type="city">Tucson</named-content></addr-line><country>United States</country></aff></contrib><contrib contrib-type="author"><name><surname>McShea</surname><given-names>Hanon</given-names></name><role specific-use="author">Author</role><aff><institution>Stanford University</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib contrib-type="author"><name><surname>Masel</surname><given-names>Joanna</given-names></name><role specific-use="author">Author</role><aff><institution>University of Arizona</institution><addr-line><named-content content-type="city">Tucson</named-content></addr-line><country>United States</country></aff></contrib></contrib-group></front-stub><body><p>The following is the authors’ response to the original reviews.</p><p>In addition to our responses to reviewer suggestions below, a minor bug in the calculation of CAIS was brought to our attention by a reader of our preprint. We have corrected this bug and rerun analyses, whose results became slightly stronger as noise was removed. While we were doing that, someone pointed out to us that our equations were almost the same as Kullback-Leibler divergence, which explains why our metric performed so well. We have made the numerically trivial (see before vs. after figure below) mathematical change to use Kullback-Leibler divergence instead, and now have a better story, with a solid basis in information theory, as to why CAIS works.</p><fig id="sa2fig1" position="float"><label>Author response image 1.</label><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-87335-sa2-fig1-v1.tif"/></fig><p>Unfortunately, we discovered a second bug that caused our PIC correction code to fail to perform the needed correction for phylogenetic confounding. The previously reported correlation between CAIS (or ENC) with body mass no longer survives PIC-correction. We have therefore removed this analysis from the manuscript. Our story now stands more on the theoretical basis of CAIS and ENC than on the post facto validation than it previously did. We now also present CAIS and ENC on a more equal footing. ENC results are slightly stronger, while CAIS has the complementary advantage of correcting for amino acid frequencies.</p><p>The work involved in these changes, as well as some of the responses to reviews below, justifies changing the second author into a co-first author, and adding an additional coauthor (Hanon McShea) who discovered the second bug.</p><disp-quote content-type="editor-comment"><p><bold>Reviewer #1 (Public Review):</bold></p><p>In this manuscript, the authors propose a new codon adaptation metric, Codon Adaptation Index of Species (CAIS), which they present as an easily obtainable proxy for effective population size. To permit between-species comparisons, they control for both amino acid frequencies and genomic GC content, which distinguishes their approach from existing ones. Having confirmed that CAIS negatively correlates with vertebrate body mass, as would be expected if small-bodied species with larger effective populations experience more efficient selection on codon usage, they then examine the relationship between CAIS and intrinsic structural disorder in proteins.</p><p>The idea of a robust species-level measure of codon adaptation is interesting. If CAIS is indeed a reliable proxy for the effectiveness of selection, it could be useful to analyze species without reliable life history- or mutation rate data (which will apply to many of the genomes becoming available in the near future).</p><p>A key question is whether CAIS, in fact, measures adaptation at the codon level. Unfortunately, CAIS is only validated indirectly by confirming a negative correlation with body mass. As a result, the observations about structural disorder are difficult to evaluate.</p></disp-quote><p>As discussed in the preamble above, we have replaced the body mass validation with a stronger theoretical basis in information theory.</p><disp-quote content-type="editor-comment"><p>A potential problem is that differences in GC between species are not independent of life history. Effective population size can drive compositional differences due to the effects of GC-biased gene conversion (gBGC). As noted by Galtier et al. (2018), genomic GC correlates negatively with body mass in mammals and birds. It would therefore be important to examine how gBGC might affect CAIS, and to what extent it could explain the relationship between CAIS and body mass.</p><p>Suppose that gBGC drives an increase in GC that is most pronounced at 3rd codon positions in highrecombination regions in small-bodied species. In this case, could observed codon usage depart more strongly from expectations calculated from overall genomic GC in small vertebrates compared to large ones? The authors also report that correcting for local intergenic GC was unsuccessful, based on the lack of a significant negative relationship with body mass (Figure 3D). In principle, this could also be consistent with local GC providing a relatively more appropriate baseline in regions with high recombination rates. Considering these scenarios would clarify what exactly CAIS is capturing.</p></disp-quote><p>Figure 3 (previously Supplementary Figures S5A and S5B) shows that CAIS is negligibly correlated with %GC (not robust to multiple comparisons correction), and ENC not at all. We believe this is evidence against the possibility brought up by the reviewer, i.e. that Ne might affect gBGC (and hence global %GC). This relationship, if present, could act as a confounding effect, but it is not present within our species dataset.</p><p>Note that we expect our genomic-GC-based codon usage expectations to reflect unchecked gBGC in an average genomic region, independently of whether that species has high or low Ne. Our working model is that non-selective forces, include gBGC as well as conventional mutation biases, vary among species, and that they rather than selection determine each species’ genome-wide %GC. By correcting for genome-wide %GC, CAIS and ENC correct for both mutation bias and gBGC, in order to isolate the effects of selection.</p><p>This argument, based on an average genomic region, is vulnerable to gene-rich genomic regions having differentially higher recombination rates and hence GC-biased gene conversion. However, we do not see the expected positive correlation between |𝐥𝐨𝐜𝐚𝐥 𝐆𝐂 - global GC| and CAIS (see new Figure 5), again suggesting that gene conversion strength is not a confounding factor acting on CAIS.</p><disp-quote content-type="editor-comment"><p>Given claims about &quot;exquisitely adapted species&quot;, the case for using CAIS as a measure of codon adaptation would also be stronger if a relationship with gene expression could be demonstrated. RSCU is expected to be higher in highly expressed genes. Is there any evidence that the equivalent GCcontrolled measure behaves similarly?</p></disp-quote><p>Correlations with gene expression are outside the scope of the current work, which is focused on producing and exploiting a single value of codon adaptation per species. It is indeed possible that our general approach of using Kullback-Leibler divergence to correct for genomic %GC could be useful in future work investigating differences among genes.</p><disp-quote content-type="editor-comment"><p>The manuscript is overall easy to follow, though some additional context may be helpful for the general reader. A more detailed discussion of how this work compares to the approach taken by Galtier et al. (2018), which accounted for GC content and gBGC when examining codon preferences, would be appropriate, for example. In addition, it would have been useful to mention past work that has attempted to explicitly quantify selection on codon usage.</p></disp-quote><p>One key difference between our work and that of Galtier et al. 2018 is that our approach does not rely on identifying specific codon preferences as a function of species. Our approach might therefore be robust to scenarios where different genes have different codon preferences (see Gingold et al. 2014 <ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1016/j.cell.2014.08.011">https://doi.org/10.1016/j.cell.2014.08.011</ext-link>). At a high level, our results are in broad agreement with those of Galtier et al., 2018, who found that gBGC affected all animal species, regardless of Ne, and who like us, found that the degree of selection on codon usage depended on Ne.</p><disp-quote content-type="editor-comment"><p><bold>Reviewer #2 (Public Review):</bold></p><p>## Summary</p><p>The goal of the authors in this study is to develop a more reliable approach for quantifying codon usage such that it is more comparable across species. Specifically, the authors wish to estimate the degree of adaptive codon usage, which is potentially a general proxy for the strength of selection at the molecular level. To this end, the authors created the Codon Adaptation Index for Species (CAIS) that controls for differences in amino acid usage and GC% across species. Using their new metric, the authors find a previously unobserved negative correlation between the overall adaptiveness of codon usage and body size across 118 vertebrates. As body size is negatively correlated with effective population size and thus the general strength of natural selection, the negative correlation between CAIS and body size is expected. The authors argue this was previously unobserved due to failures of other popular metrics such as Codon Adaptation Index (CAI) and the Effective Number of Codons (ENC) to adequately control for differences in amino acid usage and GC content across species. Most surprisingly, the authors also find a positive relationship between CAIS and the overall &quot;disorderedness&quot; of a species protein domains. As some of these results are unexpected, which is acknowledged by the authors, I think it would be particularly beneficial to work with some simulated datasets. I think CAIS has the potential to be a valuable tool for those interested in comparing codon adaptation across species in certain situations. However, I have certain theoretical concerns about CAIS as a direct proxy for the efficiency of selection <inline-formula><mml:math id="sa2m11"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:msub><mml:mi>N</mml:mi><mml:mi>e</mml:mi></mml:msub></mml:mrow></mml:mstyle></mml:math></inline-formula> when the mutation bias changes across species.</p><p>## Strengths</p><p>(1) I appreciate that the authors recognize the potential issues of comparing CAI when amino acid usage varies and correct for this in CAIS. I think this is sometimes an under-appreciated point in the codon usage literature, as CAI is a relative measure of codon usage bias (i.e. only considers synonyms). However, the strength of natural selection on codon usage can potentially vary across amino acids, such that comparing mean CAI between protein regions with different amino acid biases may result in spurious signals of statistical significance (see Cope et al. Biochemica et Biophysica Acta - Biomembranes 2018 for a clear example of this).</p></disp-quote><p>We now cite Cope et al. as an example of how amino acid composition can act as a confounding factor.</p><disp-quote content-type="editor-comment"><p>(2) The authors present numerous analysis using both ENC and mean CAI as a comparison to CAIS, helping given a sense of how CAIS corrects for some of the issues with these other metrics. I also enjoyed that they examined the previously unobserved relationship between codon usage bias and body size, which has bugged me ever since I saw Kessler and Dean 2014. The result comparing protein disorder to CAIS was particularly interesting and unexpected.</p></disp-quote><p>Unfortunately, our previous PIC correction code was buggy, and in fact the relationship with body size does not survive PIC correction (although it is strong prior to PIC correction). We have therefore removed it from the paper. However, the more novel result on protein disorder remains strong.</p><disp-quote content-type="editor-comment"><p>(3) The CAIS metric presented here is generally applicable to any species that has an annotated genome with protein-coding sequences.</p><p>## Weaknesses</p><p>(1) The main weakness of this work is that it lacks simulated data to confirm that it works as expected. This would be particularly useful for assessing the relationship between CAIS and the overall effect of protein structure disorder, which the authors acknowledge is an unexpected result. I think simulations could also allow the authors to assess how their metric performs in situations where mutation bias and natural selection act in the same direction vs. opposite directions. Additionally, although I appreciate their comparisons to ENC and mean CAI, the lack of comparison to other popular codon metrics for calculating the overall adaptiveness of a genome (e.g. dos Reis et al.'s <inline-formula><mml:math id="sa2m12"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>S</mml:mi></mml:mrow></mml:mstyle></mml:math></inline-formula> statistic, which is a function of tRNA Adaptation Index (tAI) and ENC) may be more appropriate. Even if results are similar to <inline-formula><mml:math id="sa2m13"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>S</mml:mi></mml:mrow></mml:mstyle></mml:math></inline-formula>, CAIS has a noted advantage that it doesn't require identifying tRNA gene copy numbers or abundances, which I think are generally less readily available than genomic GC% and protein-coding sequences.</p></disp-quote><p>The main limitation of dos Reis’s test in our view is that, like the better versions of CAI, it requires comparable orthologs across species. See also the discussion below re the benefits of proteome-wide approach. We now also note the advantage of not needing tRNA gene copy numbers and abundances.</p><p>Simulated datasets would be great, but we think it a nice addition rather than must-have, in particular because we are skeptical about whether our understanding of all relevant processes is good enough such that simulations would add much to our more heuristic argument along the lines of Figure 2. E.g. the complications of Gingold et al. 2014 cited above are pertinent, but incorporating them would make simulations quite involved. Instead, we now have a stronger theoretical justification for CAIS grounded in information theory. We have significantly expanded discussion of Figure 2 to give a clearer idea of the conceptual underpinnings of CAIS and ENC.</p><disp-quote content-type="editor-comment"><p>The authors mention the selection-mutation-drift equilibrium model, which underlies the basic ideas of this work (e.g. higher <inline-formula><mml:math id="sa2m14"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:msub><mml:mi>N</mml:mi><mml:mi>e</mml:mi></mml:msub></mml:mrow></mml:mstyle></mml:math></inline-formula> results in stronger selection on codon usage), but a more in-depth framing of CAIS in terms of this model is not given. I think this could be valuable, particularly in addressing the question &quot;are we really estimating what we think we're estimating?&quot;</p><p>Let's take a closer look at the formulation for RSCUS. From here on out, subscripts will only be used to denote the codon and it will be assumed that we are only considering the case of r = genome for some species s.<disp-formula id="sa2equ8"><mml:math id="sa2m15"><mml:mrow><mml:mi>R</mml:mi><mml:mi>S</mml:mi><mml:mi>C</mml:mi><mml:mi>U</mml:mi><mml:msub><mml:mi>S</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mfrac><mml:msub><mml:mi>O</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:msub><mml:mi>E</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mfrac></mml:mrow></mml:math></disp-formula></p><p>I think what the authors are attempting to do is &quot;divide out&quot; the effects of mutation bias (as given by <inline-formula><mml:math id="sa2m16"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:msub><mml:mi>E</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow></mml:mstyle></mml:math></inline-formula>), such that only the effects of natural selection remain, i.e. deviations from the expected frequency based on mutation bias alone represent adaptive codon usage. Consider Gilchrist et al. MBE 2015, which says that the expected frequency of codon i at selection-mutation-drift equilibrium in gene g for an amino acid with Na synonymous codons is<disp-formula id="sa2equ9"><mml:math id="sa2m17"><mml:mrow><mml:mstyle displaystyle="true" scriptlevel="0"><mml:msub><mml:mi>E</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>g</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mfrac><mml:msup><mml:mi>e</mml:mi><mml:mrow><mml:mo>−</mml:mo><mml:mi mathvariant="normal">Δ</mml:mi><mml:msub><mml:mi>M</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>−</mml:mo><mml:mi mathvariant="normal">Δ</mml:mi><mml:msub><mml:mi>η</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:msub><mml:mi>ϕ</mml:mi><mml:mi>g</mml:mi></mml:msub></mml:mrow></mml:msup><mml:mrow><mml:munderover><mml:mo>∑</mml:mo><mml:mrow><mml:mi>j</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:msub><mml:mi>N</mml:mi><mml:mi>a</mml:mi></mml:msub></mml:mrow></mml:munderover><mml:msup><mml:mi>e</mml:mi><mml:mrow><mml:mo>−</mml:mo><mml:mi mathvariant="normal">Δ</mml:mi><mml:msub><mml:mi>M</mml:mi><mml:mi>j</mml:mi></mml:msub><mml:mi mathvariant="normal">Δ</mml:mi><mml:msub><mml:mi>η</mml:mi><mml:mi>j</mml:mi></mml:msub><mml:msub><mml:mi>ϕ</mml:mi><mml:mi>g</mml:mi></mml:msub></mml:mrow></mml:msup></mml:mrow></mml:mfrac></mml:mstyle></mml:mrow></mml:math></disp-formula></p><p>where ∆M is the mutation bias, ∆η is the strength of selection scaled by the strength of drift, and φg is the gene expression level of gene g. In this case, ∆M and ∆η reflect the strength and direction of mutation bias and natural selection relative to a reference codon, for which ∆M,∆η = 0. Assuming the selection-mutation-drift equilibrium model is generally adequate to model of the true codon usage patterns in a genome (as I do and I think the authors do, too), the Ei,g could be considered the expected observed frequency codon i in gene g E[Oi,g].</p><p>Let’s re-write the <inline-formula><mml:math id="sa2m18"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:msub><mml:mi>E</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mfrac><mml:msub><mml:mi>p</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mrow><mml:munderover><mml:mo>∑</mml:mo><mml:mrow><mml:mi>j</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:msub><mml:mi>N</mml:mi><mml:mi>a</mml:mi></mml:msub></mml:mrow></mml:munderover><mml:msub><mml:mi>p</mml:mi><mml:mi>j</mml:mi></mml:msub></mml:mrow></mml:mfrac></mml:mrow></mml:mstyle></mml:math></inline-formula> in the form of Gilchrist et al., such that it is a function of mutation bias ∆M. For simplicity we will consider just the two codon case and assume the amino acid sequence is fixed. Assuming GC% is at equilibrium, the term gr and 1 − gr can be written as<disp-formula id="sa2equ10"><mml:math id="sa2m19"><mml:mrow><mml:msub><mml:mi>g</mml:mi><mml:mi>r</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mfrac><mml:msub><mml:mi>μ</mml:mi><mml:mrow><mml:mi>A</mml:mi><mml:mi>T</mml:mi><mml:mo stretchy="false">→</mml:mo><mml:mi>G</mml:mi><mml:mi>C</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:msub><mml:mi>μ</mml:mi><mml:mrow><mml:mi>A</mml:mi><mml:mi>T</mml:mi><mml:mo stretchy="false">→</mml:mo><mml:mi>G</mml:mi><mml:mi>C</mml:mi></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mi>μ</mml:mi><mml:mrow><mml:mi>G</mml:mi><mml:mi>C</mml:mi><mml:mo stretchy="false">→</mml:mo><mml:mi>A</mml:mi><mml:mi>T</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfrac><mml:mn>1</mml:mn><mml:mo>−</mml:mo><mml:msub><mml:mi>g</mml:mi><mml:mi>r</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mfrac><mml:msub><mml:mi>μ</mml:mi><mml:mrow><mml:mi>G</mml:mi><mml:mi>C</mml:mi><mml:mo stretchy="false">→</mml:mo><mml:mi>A</mml:mi><mml:mi>T</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:msub><mml:mi>μ</mml:mi><mml:mrow><mml:mi>A</mml:mi><mml:mi>T</mml:mi><mml:mo stretchy="false">→</mml:mo><mml:mi>G</mml:mi><mml:mi>C</mml:mi></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mi>μ</mml:mi><mml:mrow><mml:mi>G</mml:mi><mml:mi>C</mml:mi><mml:mo stretchy="false">→</mml:mo><mml:mi>A</mml:mi><mml:mi>T</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfrac></mml:mrow></mml:math></disp-formula></p><p>where µx→y is the mutation rate from nucleotides x to y. As described in Gilchrist et al. MBE 2015 and Shah and Gilchrist PNAS 2011, the mutation bias <inline-formula><mml:math id="sa2m20"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi mathvariant="normal">Δ</mml:mi><mml:msub><mml:mi>M</mml:mi><mml:mrow><mml:mi>N</mml:mi><mml:mi>N</mml:mi><mml:mi>A</mml:mi><mml:mo>,</mml:mo><mml:mi>N</mml:mi><mml:mi>N</mml:mi><mml:mi>G</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mi>l</mml:mi><mml:mi>o</mml:mi><mml:mi>g</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mfrac><mml:msub><mml:mi>μ</mml:mi><mml:mrow><mml:mi>A</mml:mi><mml:mi>T</mml:mi><mml:mo stretchy="false">→</mml:mo><mml:mi>G</mml:mi><mml:mi>C</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mi>μ</mml:mi><mml:mrow><mml:mi>G</mml:mi><mml:mi>C</mml:mi><mml:mo stretchy="false">→</mml:mo><mml:mi>A</mml:mi><mml:mi>T</mml:mi></mml:mrow></mml:msub></mml:mfrac><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mstyle></mml:math></inline-formula> .This can be expressed in terms of the equilibrium GC content by recognizing that<disp-formula id="sa2equ11"><mml:math id="sa2m21"><mml:mrow><mml:mfrac><mml:msub><mml:mi>g</mml:mi><mml:mi>r</mml:mi></mml:msub><mml:mrow><mml:mn>1</mml:mn><mml:mo>−</mml:mo><mml:msub><mml:mi>g</mml:mi><mml:mi>r</mml:mi></mml:msub></mml:mrow></mml:mfrac><mml:mo>=</mml:mo><mml:mfrac><mml:msub><mml:mi>μ</mml:mi><mml:mrow><mml:mi>A</mml:mi><mml:mi>T</mml:mi><mml:mo stretchy="false">→</mml:mo><mml:mi>G</mml:mi><mml:mi>C</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mi>μ</mml:mi><mml:mrow><mml:mi>G</mml:mi><mml:mi>C</mml:mi><mml:mo stretchy="false">→</mml:mo><mml:mi>A</mml:mi><mml:mi>T</mml:mi></mml:mrow></mml:msub></mml:mfrac><mml:mspace linebreak="newline"/><mml:mspace linebreak="newline"/><mml:mi>i</mml:mi><mml:mi>m</mml:mi><mml:mi>p</mml:mi><mml:mi>l</mml:mi><mml:mi>i</mml:mi><mml:mi>e</mml:mi><mml:mi>s</mml:mi><mml:mfrac><mml:msub><mml:mi>g</mml:mi><mml:mi>r</mml:mi></mml:msub><mml:mrow><mml:mn>1</mml:mn><mml:mo>−</mml:mo><mml:msub><mml:mi>g</mml:mi><mml:mi>r</mml:mi></mml:msub></mml:mrow></mml:mfrac><mml:mo>=</mml:mo><mml:msup><mml:mi>e</mml:mi><mml:mrow><mml:mi mathvariant="normal">Δ</mml:mi><mml:mi>M</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:math></disp-formula></p><p>As we are assuming the amino acid sequence is fixed, the probability of observing a synonymous codon i at an amino acid becomes just a Bernoulli process.<disp-formula id="sa2equ12"><mml:math id="sa2m22"><mml:mrow><mml:msub><mml:mi>p</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:msubsup><mml:mi>g</mml:mi><mml:mi>r</mml:mi><mml:mi>x</mml:mi></mml:msubsup><mml:mo stretchy="false">(</mml:mo><mml:mn>1</mml:mn><mml:mo>−</mml:mo><mml:msub><mml:mi>g</mml:mi><mml:mi>r</mml:mi></mml:msub><mml:msup><mml:mo stretchy="false">)</mml:mo><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mn>1</mml:mn><mml:mo>−</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:msup></mml:mrow></mml:math></disp-formula></p><p>If we do this, then<disp-formula id="sa2equ13"><mml:math id="sa2m23"><mml:mrow><mml:msub><mml:mi>E</mml:mi><mml:mrow><mml:mi>N</mml:mi><mml:mi>N</mml:mi><mml:mi>A</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mfrac><mml:msub><mml:mi>p</mml:mi><mml:mrow><mml:mi>N</mml:mi><mml:mi>N</mml:mi><mml:mi>A</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:msub><mml:mi>p</mml:mi><mml:mrow><mml:mi>N</mml:mi><mml:mi>N</mml:mi><mml:mi>A</mml:mi></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mi>p</mml:mi><mml:mrow><mml:mi>N</mml:mi><mml:mi>N</mml:mi><mml:mi>G</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfrac><mml:mtext> </mml:mtext><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn><mml:mo>−</mml:mo><mml:msub><mml:mi>g</mml:mi><mml:mi>r</mml:mi></mml:msub></mml:mrow><mml:mrow><mml:msub><mml:mi>g</mml:mi><mml:mi>r</mml:mi></mml:msub><mml:mo>+</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:mn>1</mml:mn><mml:mo>−</mml:mo><mml:msub><mml:mi>g</mml:mi><mml:mi>r</mml:mi></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mfrac><mml:mtext> </mml:mtext><mml:mo>=</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mrow><mml:mfrac><mml:msub><mml:mi>g</mml:mi><mml:mi>r</mml:mi></mml:msub><mml:mrow><mml:mn>1</mml:mn><mml:mo>−</mml:mo><mml:msub><mml:mi>g</mml:mi><mml:mi>r</mml:mi></mml:msub></mml:mrow></mml:mfrac><mml:mo>+</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:mfrac><mml:mtext> </mml:mtext><mml:mo>=</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mrow><mml:msup><mml:mi>e</mml:mi><mml:mrow><mml:mi mathvariant="normal">Δ</mml:mi><mml:mi>M</mml:mi></mml:mrow></mml:msup><mml:mo>+</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:mfrac><mml:mtext> </mml:mtext><mml:mo>=</mml:mo><mml:mfrac><mml:msup><mml:mi>e</mml:mi><mml:mrow><mml:mo>−</mml:mo><mml:mi mathvariant="normal">Δ</mml:mi><mml:mi>M</mml:mi></mml:mrow></mml:msup><mml:mrow><mml:mn>1</mml:mn><mml:mo>+</mml:mo><mml:msup><mml:mi>e</mml:mi><mml:mrow><mml:mo>−</mml:mo><mml:mi mathvariant="normal">Δ</mml:mi><mml:mi>M</mml:mi></mml:mrow></mml:msup></mml:mrow></mml:mfrac></mml:mrow></mml:math></disp-formula></p><p>Recall that in the Gilchrist et al. framework, the reference codon has ∆MNNG,NNG = 0 = ⇒ e−∆MNNG,NNG = 1. Thus, we have recovered the Gilchrist et al. model from the formulation of <inline-formula><mml:math id="sa2m24"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:msub><mml:mi>E</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow></mml:mstyle></mml:math></inline-formula> under the assumption that natural selection has no impact on codon usage and codon NNG is the pre-defined reference codon. To see this, plug in 0 for ∆η in equation (1)..</p><p>We can then calculate the expected RSCUS using equation (1) (using notation E[Oi]) and equation (6) for the two codon case. For simplicity assume, we are only considering a gene of average expression (defined as <inline-formula><mml:math id="sa2m25"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:msub><mml:mi>ϕ</mml:mi><mml:mi>g</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:mstyle></mml:math></inline-formula>). Assume in this case that NNG is the reference codon (∆MNNG,∆ηNNG = 0).<disp-formula id="sa2equ14"><mml:math id="sa2m26"><mml:mrow><mml:mi>E</mml:mi><mml:mo stretchy="false">[</mml:mo><mml:mi>R</mml:mi><mml:mi>S</mml:mi><mml:mi>C</mml:mi><mml:mi>U</mml:mi><mml:msub><mml:mi>S</mml:mi><mml:mrow><mml:mi>N</mml:mi><mml:mi>N</mml:mi><mml:mi>A</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">]</mml:mo><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>E</mml:mi><mml:mo stretchy="false">[</mml:mo><mml:msub><mml:mi>O</mml:mi><mml:mrow><mml:mi>N</mml:mi><mml:mi>N</mml:mi><mml:mi>A</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">]</mml:mo></mml:mrow><mml:msub><mml:mi>E</mml:mi><mml:mrow><mml:mi>N</mml:mi><mml:mi>N</mml:mi><mml:mi>A</mml:mi></mml:mrow></mml:msub></mml:mfrac><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:msup><mml:mi>e</mml:mi><mml:mrow><mml:mo>−</mml:mo><mml:mi mathvariant="normal">Δ</mml:mi><mml:msub><mml:mi>η</mml:mi><mml:mrow><mml:mi>N</mml:mi><mml:mi>N</mml:mi><mml:mi>A</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msup><mml:mo stretchy="false">(</mml:mo><mml:msup><mml:mi>e</mml:mi><mml:mrow><mml:mo>−</mml:mo><mml:mi mathvariant="normal">Δ</mml:mi><mml:msub><mml:mi>M</mml:mi><mml:mrow><mml:mi>N</mml:mi><mml:mi>N</mml:mi><mml:mi>A</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msup><mml:mo>+</mml:mo><mml:msup><mml:mi>e</mml:mi><mml:mrow><mml:mo>−</mml:mo><mml:mi mathvariant="normal">Δ</mml:mi><mml:msub><mml:mi>M</mml:mi><mml:mrow><mml:mi>N</mml:mi><mml:mi>N</mml:mi><mml:mi>G</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msup><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mrow><mml:msup><mml:mi>e</mml:mi><mml:mrow><mml:mo>−</mml:mo><mml:mi mathvariant="normal">Δ</mml:mi><mml:msub><mml:mi>M</mml:mi><mml:mrow><mml:mi>N</mml:mi><mml:mi>N</mml:mi><mml:mi>A</mml:mi></mml:mrow></mml:msub><mml:mo>−</mml:mo><mml:mi mathvariant="normal">Δ</mml:mi><mml:msub><mml:mi>η</mml:mi><mml:mrow><mml:mi>N</mml:mi><mml:mi>N</mml:mi><mml:mi>A</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msup><mml:mo>+</mml:mo><mml:msup><mml:mi>e</mml:mi><mml:mrow><mml:mo>−</mml:mo><mml:mi mathvariant="normal">Δ</mml:mi><mml:msub><mml:mi>M</mml:mi><mml:mrow><mml:mi>N</mml:mi><mml:mi>N</mml:mi><mml:mi>G</mml:mi></mml:mrow></mml:msub><mml:mi mathvariant="normal">Δ</mml:mi><mml:msub><mml:mi>η</mml:mi><mml:mrow><mml:mi>N</mml:mi><mml:mi>N</mml:mi><mml:mi>G</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msup></mml:mrow></mml:mfrac><mml:mspace linebreak="newline"/><mml:mspace linebreak="newline"/><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:msup><mml:mi>e</mml:mi><mml:mrow><mml:mo>−</mml:mo><mml:mi mathvariant="normal">Δ</mml:mi><mml:msub><mml:mi>M</mml:mi><mml:mrow><mml:mi>N</mml:mi><mml:mi>N</mml:mi><mml:mi>A</mml:mi></mml:mrow></mml:msub><mml:mo>−</mml:mo><mml:mi mathvariant="normal">Δ</mml:mi><mml:msub><mml:mi>η</mml:mi><mml:mrow><mml:mi>N</mml:mi><mml:mi>N</mml:mi><mml:mi>A</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msup><mml:mo>+</mml:mo><mml:msup><mml:mi>e</mml:mi><mml:mrow><mml:mo>−</mml:mo><mml:mi mathvariant="normal">Δ</mml:mi><mml:msub><mml:mi>M</mml:mi><mml:mrow><mml:mi>N</mml:mi><mml:mi>N</mml:mi><mml:mi>G</mml:mi></mml:mrow></mml:msub><mml:mo>−</mml:mo><mml:mi mathvariant="normal">Δ</mml:mi><mml:msub><mml:mi>η</mml:mi><mml:mrow><mml:mi>N</mml:mi><mml:mi>N</mml:mi><mml:mi>A</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msup></mml:mrow><mml:mrow><mml:msup><mml:mi>e</mml:mi><mml:mrow><mml:mo>−</mml:mo><mml:mi mathvariant="normal">Δ</mml:mi><mml:msub><mml:mi>M</mml:mi><mml:mrow><mml:mi>N</mml:mi><mml:mi>N</mml:mi><mml:mi>A</mml:mi></mml:mrow></mml:msub><mml:mo>−</mml:mo><mml:mi mathvariant="normal">Δ</mml:mi><mml:msub><mml:mi>η</mml:mi><mml:mrow><mml:mi>N</mml:mi><mml:mi>N</mml:mi><mml:mi>A</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msup><mml:mo>+</mml:mo><mml:msup><mml:mi>e</mml:mi><mml:mrow><mml:mo>−</mml:mo><mml:mi mathvariant="normal">Δ</mml:mi><mml:msub><mml:mi>M</mml:mi><mml:mrow><mml:mi>N</mml:mi><mml:mi>N</mml:mi><mml:mi>G</mml:mi></mml:mrow></mml:msub><mml:mo>−</mml:mo><mml:mi mathvariant="normal">Δ</mml:mi><mml:msub><mml:mi>η</mml:mi><mml:mrow><mml:mi>N</mml:mi><mml:mi>N</mml:mi><mml:mi>G</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msup></mml:mrow></mml:mfrac><mml:mspace linebreak="newline"/><mml:mspace linebreak="newline"/><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:msup><mml:mi>e</mml:mi><mml:mrow><mml:mo>−</mml:mo><mml:mi mathvariant="normal">Δ</mml:mi><mml:msub><mml:mi>M</mml:mi><mml:mrow><mml:mi>N</mml:mi><mml:mi>N</mml:mi><mml:mi>A</mml:mi></mml:mrow></mml:msub><mml:mo>−</mml:mo><mml:mi mathvariant="normal">Δ</mml:mi><mml:msub><mml:mi>η</mml:mi><mml:mrow><mml:mi>N</mml:mi><mml:mi>N</mml:mi><mml:mi>A</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msup><mml:mo>+</mml:mo><mml:msup><mml:mi>e</mml:mi><mml:mrow><mml:mo>−</mml:mo><mml:mi mathvariant="normal">Δ</mml:mi><mml:msub><mml:mi>η</mml:mi><mml:mrow><mml:mi>N</mml:mi><mml:mi>N</mml:mi><mml:mi>A</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msup></mml:mrow><mml:mrow><mml:msup><mml:mi>e</mml:mi><mml:mrow><mml:mi mathvariant="normal">Δ</mml:mi><mml:msub><mml:mi>M</mml:mi><mml:mrow><mml:mi>N</mml:mi><mml:mi>N</mml:mi><mml:mi>A</mml:mi></mml:mrow></mml:msub><mml:mo>−</mml:mo><mml:mi mathvariant="normal">Δ</mml:mi><mml:msub><mml:mi>η</mml:mi><mml:mrow><mml:mi>N</mml:mi><mml:mi>N</mml:mi><mml:mi>A</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msup><mml:mo>+</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:mfrac></mml:mrow></mml:math></disp-formula></p><p>This shows that the expected value of RSCUS for a two-codon amino acid is expected to increase as the strength of selection <inline-formula><mml:math id="sa2m27"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi mathvariant="normal">Δ</mml:mi><mml:mi>η</mml:mi></mml:mrow></mml:mstyle></mml:math></inline-formula> increases, which is desired. Note that <inline-formula><mml:math id="sa2m28"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi mathvariant="normal">Δ</mml:mi><mml:mi>η</mml:mi></mml:mrow></mml:mstyle></mml:math></inline-formula> in Gilchrist et al. is formulated in terms of selection *against* a codon relative to the reference, such that a negative value represents that a codon is favored relative to the reference. If (i.e. selection does not favor either codon), then <inline-formula><mml:math id="sa2m29"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>E</mml:mi><mml:mo stretchy="false">[</mml:mo><mml:mi>R</mml:mi><mml:mi>S</mml:mi><mml:mi>C</mml:mi><mml:mi>U</mml:mi><mml:mi>S</mml:mi><mml:mo stretchy="false">]</mml:mo><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:mstyle></mml:math></inline-formula>. Also note that the expected RSCUS does not remain independent of the mutation bias. This means that even if <inline-formula><mml:math id="sa2m30"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>s</mml:mi><mml:msub><mml:mi>N</mml:mi><mml:mi>e</mml:mi></mml:msub></mml:mrow></mml:mstyle></mml:math></inline-formula> (i.e. the strength of natural selection) does not change between species, changes to the strength and direction of mutation bias across species could impact RSCUS. Assuming my math is right, I think one needs to be cautious when interpreting CAIS as representative of the differences in the efficiency of selection across species except under very particular circumstances. One such case could be when it is known that mutation bias varies little across the species of interest. Looking at the species used in this manuscript, most of them have a GC content ranging around 0.41, so I suspect their results are okay.</p><p>Although I have not done so, I am sure this could be extended to the 4 and 6 codon amino acids.</p></disp-quote><p>We thank Reviewer 2 for explicitly laying out the math that was implicit in our Figures 1 and 2. While we keep our more heuristic presentation, our revised manuscript now more clearly acknowledges that the per-site codon adaptation bias depicted in Figure 1 has limited sensitivity to s*Ne. The reason that we believe our approach worked despite this, is that we think the phenomenon is driven by what is shown in Figure 2. I.e., where Ne makes a difference is by determining the proteome-wide fraction of codons subject to significant codon adaptation, rather than by determining the strength of codon adaptation at any particular site or gene. We have made multiple changes to the texts to make this point clearer.</p><disp-quote content-type="editor-comment"><p>Another minor weakness of this work is that although the method is generally applicable to any species with an annotated genome and the code is publicly available, the code itself contains hard-coded values for GC% and amino acid frequencies across the 118 vertebrates. The lack of a more flexible tool may make it difficult for less computationally-experienced researchers to take advantage of this method.</p></disp-quote><p>Genome-wide %GC values are hard-coded because they were taken from the previous study of James et al. (2023) https://doi.org/10.1093/molbev/msad073. As summarized in the manuscript, genome-wide %GC was a byproduct of a scan of all six reading frames across genic and intergenic sequences available from NCBI with access dates between May and July 2019. The more complicated code used to calculate the intergenic %GC, and the code used to calculate amino acid frequencies is located at https://github.com/MaselLab/CodonAdaptation-Index-of-Species. Luckily, someone else just wrote a simpler end to end pipeline for us, on the basis of our preprint. We now note this in the Acknowledgements, and link to it: https://github.com/gavinmdouglas/handy_pop_gen/blob/main/CAIS.py.</p></body></sub-article></article>