<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Archiving and Interchange DTD with MathML3 v1.3 20210610//EN"  "JATS-archivearticle1-3-mathml3.dtd"><article xmlns:ali="http://www.niso.org/schemas/ali/1.0/" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article" dtd-version="1.3"><front><journal-meta><journal-id journal-id-type="nlm-ta">elife</journal-id><journal-id journal-id-type="publisher-id">eLife</journal-id><journal-title-group><journal-title>eLife</journal-title></journal-title-group><issn publication-format="electronic" pub-type="epub">2050-084X</issn><publisher><publisher-name>eLife Sciences Publications, Ltd</publisher-name></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">89656</article-id><article-id pub-id-type="doi">10.7554/eLife.89656</article-id><article-id pub-id-type="doi" specific-use="version">10.7554/eLife.89656.3</article-id><article-categories><subj-group subj-group-type="display-channel"><subject>Research Article</subject></subj-group><subj-group subj-group-type="heading"><subject>Biochemistry and Chemical Biology</subject></subj-group><subj-group subj-group-type="heading"><subject>Computational and Systems Biology</subject></subj-group></article-categories><title-group><article-title>Enrichment of rare codons at 5' ends of genes is a spandrel caused by evolutionary sequence turnover and does not improve translation</article-title></title-group><contrib-group><contrib contrib-type="author" id="author-320708"><name><surname>Sejour</surname><given-names>Richard</given-names></name><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="fn" rid="con1"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author" id="author-20811"><name><surname>Leatherwood</surname><given-names>Janet</given-names></name><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="fn" rid="con2"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author" id="author-3817"><name><surname>Yurovsky</surname><given-names>Alisa</given-names></name><xref ref-type="aff" rid="aff3">3</xref><xref ref-type="fn" rid="con3"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author" corresp="yes" id="author-15353"><name><surname>Futcher</surname><given-names>Bruce</given-names></name><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0002-1012-9022</contrib-id><email>bfutcher@gmail.com</email><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="other" rid="fund1"/><xref ref-type="other" rid="fund2"/><xref ref-type="fn" rid="con4"/><xref ref-type="fn" rid="conf1"/></contrib><aff id="aff1"><label>1</label><institution-wrap><institution-id institution-id-type="ror">https://ror.org/05qghxh33</institution-id><institution>Department of Pharmacological Sciences, Stony Brook University</institution></institution-wrap><addr-line><named-content content-type="city">Stony Brook</named-content></addr-line><country>United States</country></aff><aff id="aff2"><label>2</label><institution-wrap><institution-id institution-id-type="ror">https://ror.org/05qghxh33</institution-id><institution>Department of Microbiology and Immunology, Stony Brook University</institution></institution-wrap><addr-line><named-content content-type="city">Stony Brook</named-content></addr-line><country>United States</country></aff><aff id="aff3"><label>3</label><institution-wrap><institution-id institution-id-type="ror">https://ror.org/05qghxh33</institution-id><institution>Department of Biomedical Informatics, Stony Brook University</institution></institution-wrap><addr-line><named-content content-type="city">Stony Brook</named-content></addr-line><country>United States</country></aff></contrib-group><contrib-group content-type="section"><contrib contrib-type="editor"><name><surname>Hinnebusch</surname><given-names>Alan G</given-names></name><role>Reviewing Editor</role><aff><institution-wrap><institution-id institution-id-type="ror">https://ror.org/04byxyr05</institution-id><institution>Eunice Kennedy Shriver National Institute of Child Health and Human Development</institution></institution-wrap><country>United States</country></aff></contrib><contrib contrib-type="senior_editor"><name><surname>Weigel</surname><given-names>Detlef</given-names></name><role>Senior Editor</role><aff><institution-wrap><institution-id institution-id-type="ror">https://ror.org/0243gzr89</institution-id><institution>Max Planck Institute for Biology Tübingen</institution></institution-wrap><country>Germany</country></aff></contrib></contrib-group><pub-date publication-format="electronic" date-type="publication"><day>15</day><month>07</month><year>2024</year></pub-date><volume>12</volume><elocation-id>RP89656</elocation-id><history><date date-type="sent-for-review" iso-8601-date="2023-06-02"><day>02</day><month>06</month><year>2023</year></date></history><pub-history><event><event-desc>This manuscript was published as a preprint.</event-desc><date date-type="preprint" iso-8601-date="2023-07-11"><day>11</day><month>07</month><year>2023</year></date><self-uri content-type="preprint" xlink:href="https://doi.org/10.1101/2022.06.27.497802"/></event><event><event-desc>This manuscript was published as a reviewed preprint.</event-desc><date date-type="reviewed-preprint" iso-8601-date="2023-07-27"><day>27</day><month>07</month><year>2023</year></date><self-uri content-type="reviewed-preprint" xlink:href="https://doi.org/10.7554/eLife.89656.1"/></event><event><event-desc>The reviewed preprint was revised.</event-desc><date date-type="reviewed-preprint" iso-8601-date="2023-11-13"><day>13</day><month>11</month><year>2023</year></date><self-uri content-type="reviewed-preprint" xlink:href="https://doi.org/10.7554/eLife.89656.2"/></event></pub-history><permissions><copyright-statement>© 2023, Sejour et al</copyright-statement><copyright-year>2023</copyright-year><copyright-holder>Sejour et al</copyright-holder><ali:free_to_read/><license xlink:href="http://creativecommons.org/licenses/by/4.0/"><ali:license_ref>http://creativecommons.org/licenses/by/4.0/</ali:license_ref><license-p>This article is distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="http://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution License</ext-link>, which permits unrestricted use and redistribution provided that the original author and source are credited.</license-p></license></permissions><self-uri content-type="pdf" xlink:href="elife-89656-v1.pdf"/><self-uri content-type="figures-pdf" xlink:href="elife-89656-figures-v1.pdf"/><abstract><p>Previously, Tuller et al. found that the first 30–50 codons of the genes of yeast and other eukaryotes are slightly enriched for rare codons. They argued that this slowed translation, and was adaptive because it queued ribosomes to prevent collisions. Today, the translational speeds of different codons are known, and indeed rare codons are translated slowly. We re-examined this 5’ slow translation ‘ramp.’ We confirm that 5’ regions are slightly enriched for rare codons; in addition, they are depleted for downstream Start codons (which are fast), with both effects contributing to slow 5’ translation. However, we also find that the 5’ (and 3’) ends of yeast genes are poorly conserved in evolution, suggesting that they are unstable and turnover relatively rapidly. When a new 5’ end forms de novo, it is likely to include codons that would otherwise be rare. Because evolution has had a relatively short time to select against these codons, 5’ ends are typically slightly enriched for rare, slow codons. Opposite to the expectation of Tuller et al., we show by direct experiment that genes with slowly translated codons at the 5’ end are expressed relatively poorly, and that substituting faster synonymous codons improves expression. Direct experiment shows that slow codons do not prevent downstream ribosome collisions. Further informatic studies suggest that for natural genes, slow 5’ ends are correlated with poor gene expression, opposite to the expectation of Tuller et al. Thus, we conclude that slow 5’ translation is a ‘spandrel’--a non-adaptive consequence of something else, in this case, the turnover of 5’ ends in evolution, and it does not improve translation.</p></abstract><kwd-group kwd-group-type="author-keywords"><kwd>translation</kwd><kwd>codon usage</kwd><kwd>translation speed</kwd><kwd>ribosome collisions</kwd></kwd-group><kwd-group kwd-group-type="research-organism"><title>Research organism</title><kwd><italic>S. cerevisiae</italic></kwd></kwd-group><funding-group><award-group id="fund1"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100000002</institution-id><institution>National Institutes of Health</institution></institution-wrap></funding-source><award-id>RO1 GM127542</award-id><principal-award-recipient><name><surname>Futcher</surname><given-names>Bruce</given-names></name></principal-award-recipient></award-group><award-group id="fund2"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100000002</institution-id><institution>National Institutes of Health</institution></institution-wrap></funding-source><award-id>RO1 GM 132238</award-id><principal-award-recipient><name><surname>Futcher</surname><given-names>Bruce</given-names></name></principal-award-recipient></award-group><funding-statement>The funders had no role in study design, data collection and interpretation, or the decision to submit the work for publication.</funding-statement></funding-group><custom-meta-group><custom-meta specific-use="meta-only"><meta-name>Author impact statement</meta-name><meta-value>The 5' ends of genes are slightly enriched for rare codons largely because the ends turnover in evolution and gather rare codons, and these rare codons do not improve translation.</meta-value></custom-meta><custom-meta specific-use="meta-only"><meta-name>publishing-route</meta-name><meta-value>prc</meta-value></custom-meta></custom-meta-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>(<xref ref-type="bibr" rid="bib32">Tuller et al., 2010</xref>) were interested in the idea that a slow translational ramp at the beginning of a gene might queue ribosomes in an orderly way, thereby preventing ribosome traffic jams and collisions. However, at that time, translation speeds for the 61 sense codons were not known from direct measurement. Therefore, as a proxy for codon translation speed, Tuller et al. devised a proxy speed measurement based on the tRNA-adaptation index (tAI), a measure of the abundance of each tRNA. The assumption is that codons recognized by more abundant tRNAs would be translated faster. Using this proxy, Tuller et al. found that in yeast and other eukaryotes, the first 30–100 codons are enriched for codons for which tRNAs are rare (generally, rare codons), and are presumably translated slowly. The size of the effect is small (about a 3% difference, Figure 2C of <xref ref-type="bibr" rid="bib32">Tuller et al., 2010</xref>), but is statistically highly significant.</p><p>At the time of the work of Tuller et al. ribosome profiling had recently been developed (<xref ref-type="bibr" rid="bib19">Ingolia et al., 2009</xref>), and early ribosome profiling showed a high density of ribosomes near the 5’ end of the mRNA, consistent with slow translation in this region, and the rare codons found by Tuller et al. could have contributed to this. However, later work showed that this 5’ high density of ribosomes was an artifact of the way cycloheximide was used to arrest translation in the original protocol (<xref ref-type="bibr" rid="bib36">Weinberg et al., 2016</xref>). With newer protocols for ribosome profiling, which use cycloheximide only at later steps, the region of 5’ high ribosome density largely, but not entirely, disappears (<xref ref-type="bibr" rid="bib36">Weinberg et al., 2016</xref>) (see Discussion).</p><p>Since then, many workers have used ribosome profiling to directly measure the speed of translation of individual codons (cited below). With such data in hand, we revisited the issues addressed by <xref ref-type="bibr" rid="bib32">Tuller et al., 2010</xref>. On the one hand, our analyses confirm that the 5’ regions of genes are typically slightly enriched for rare codons, and these encodings likely slow translation. On the other hand, various aspects of the data led us to an alternative hypothesis, namely that the 5’ ends were turning over relatively rapidly in evolution; that these 5’ ends were, therefore, relatively young; and that selection had not yet succeeded in removing all the rare, slow codons initially present in the de novo 5’ ends. We did a direct experimental test of the effects of slow or fast initial translation. Opposite to Tuller et al., we found that encoding slow initial translation resulted in lower protein production than fast initial translation. This continued to be true even when we placed ribosome collision sites inside the reporter gene. Thus a slow initial translation ramp, though present, neither improves gene expression nor prevents ribosome collisions.</p><p>It is natural to assume that the enrichment of slow codons near 5’ ends is a product of selection. However, as elegantly argued by <xref ref-type="bibr" rid="bib15">Gould and Lewontin, 1979</xref> in their classic paper ‘The Spandrels of San Marco and the Panglossian Paradigm: A Critique of the Adaptionist Programme,’ not all biological phenomena are adaptive, or even a direct product of selection. They argued from the example of a ‘spandrel:’ in architecture, a triangular space created when an arch supports a lintel. There is no architectural role for spandrels as such; they are the indirect and inevitable result of the juxtaposition of two other functional architectural elements. We argue that the slightly slow initial translation of eukaryotic genes may likewise be a spandrel, a non-adaptive consequence of something else, the instability of 5’ ends in evolution.</p></sec><sec id="s2" sec-type="results"><title>Results</title><sec id="s2-1"><title>Calculations of encoded translation speed imply slow initial translation</title><p><xref ref-type="bibr" rid="bib32">Tuller et al., 2010</xref> used codon-specific tRNA abundance as a proxy to estimate the speed of translation of codons. Since then, analysis of ribosome profiling data has yielded direct measurements of the translation speed of individual codons (<xref ref-type="bibr" rid="bib36">Weinberg et al., 2016</xref>; <xref ref-type="bibr" rid="bib9">Dao Duc and Song, 2018</xref>; <xref ref-type="bibr" rid="bib14">Gardin et al., 2014</xref>; <xref ref-type="bibr" rid="bib16">Gritsenko et al., 2015</xref>; <xref ref-type="bibr" rid="bib23">Lareau et al., 2014</xref>; <xref ref-type="bibr" rid="bib30">Sharma et al., 2019</xref>; <xref ref-type="bibr" rid="bib33">Tunney et al., 2018</xref>; <xref ref-type="bibr" rid="bib35">Wang et al., 2017</xref>). Accordingly, we have repeated some of the work of Tuller et al. but using the Ribosome Residence Time (RRT, Methods and materials, <xref ref-type="supplementary-material" rid="supp1">Supplementary file 1</xref>; <xref ref-type="bibr" rid="bib14">Gardin et al., 2014</xref>) of each of the 61 sense codons as a measure of translation speed. We refer to ‘encoded translation speed’ to specify that we are focusing purely on the effects of different codons on translation speed, and not on other factors that might differentially affect translation speed at different regions of the mRNA, such as secondary structure.</p><p>Using the RRT, encoded translation speeds were calculated in sliding windows across all coding ORFs from <italic>S. cerevisiae</italic>. The start codon was omitted because it is constant across genes and is an unusually ‘fast’ codon. Consistent with Tuller et al. we find that the first 30–100 codons had lower calculated translation speeds than the rest of the gene (<xref ref-type="fig" rid="fig1">Figure 1</xref>). We focused our analyses on the first 40 codons for comparability to Tuller et al.; we call this the ‘Slow Initial Translation’ region, or SIT. Although the tendency towards slow translation near the beginning of the gene is very highly statistically significant, the size of the effect is small. When we compare the first 40 codons of a gene to the rest of the same gene, we find that the average difference in encoded translation speed is about 1.2% (similar to <xref ref-type="bibr" rid="bib32">Tuller et al., 2010</xref>), with a p-value &lt;0.001. For comparison, codons can vary in translation speed by about threefold (<xref ref-type="supplementary-material" rid="supp1">Supplementary file 1</xref>), or perhaps as much as sixfold (<xref ref-type="bibr" rid="bib36">Weinberg et al., 2016</xref>, their Table S2).</p><fig-group><fig id="fig1" position="float"><label>Figure 1.</label><caption><title>Calculation of translation speed confirms slow initial translation (SIT).</title><p>Translation speeds were calculated using ribosome residence time (RRT) (<xref ref-type="bibr" rid="bib14">Gardin et al., 2014</xref>; <xref ref-type="supplementary-material" rid="supp1">Supplementary file 1</xref> for RRT values) as a measure of codon-specific translation speed over <italic>S. cerevisiae</italic> open reading frames (ORFs). The horizontal line indicates average inverse RRT across all ORFs. The average speed in the first 40 amino acids is about 1.1% slower than in the rest of the gene (p &lt; 0.001).</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-89656-fig1-v1.tif"/></fig><fig id="fig1s1" position="float" specific-use="child-fig"><label>Figure 1—figure supplement 1.</label><caption><title>Distribution of translation speeds at 5’ and 3’ ends.</title><p>The distribution of relative translation speeds over 5694 genes is shown for the first and last 40 amino acids. For the 5’ end, 57.2% of genes have relatively slow initial translation, while for the 3’ end, 50.14% of genes have a slow terminal translation.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-89656-fig1-figsupp1-v1.tif"/></fig></fig-group><p>Although on average genes are translated slowly near their 5’ ends, there is variability. <xref ref-type="fig" rid="fig1s1">Figure 1—figure supplement 1</xref> shows the distribution of initial encoded translation speed for all yeast ORFs. About 57% of genes have a slow initial translation (SIT) region, while the remainder have fast initial translation (FIT).</p></sec><sec id="s2-2"><title>Rare (slow) codons are enriched within the first 40 codons</title><p>Rare codons tend to be slow codons, and <italic>vice versa</italic> (<xref ref-type="bibr" rid="bib14">Gardin et al., 2014</xref>; <xref ref-type="fig" rid="fig2s1">Figure 2—figure supplement 1</xref>). But the correlation is not perfect—other things being equal, A/T-rich codons tend to be faster than G/C-rich codons, and codons with a third position wobble base tend to be faster than codons with the cognate base (<xref ref-type="bibr" rid="bib14">Gardin et al., 2014</xref>). To better understand why gene beginnings are more slowly translated, we examined the relative usage of each of the 61 sense codons in the first 40 codons after but not including the initiator ATG (<xref ref-type="fig" rid="fig2">Figure 2</xref>).</p><fig-group><fig id="fig2" position="float"><label>Figure 2.</label><caption><title>Codon usage in the slow initial translation (SIT) region.</title><p>(<bold>A</bold>). Relative codon usage in the SIT versus the rest of the gene. The Y-axis shows codon usage in the first 40 amino acids (omitting ATG) divided by its usage in the rest of the gene. The 61 sense codons are grouped by amino acid. Within each group, codons are ordered from least to most frequent left to right. Red arrows show the seven slowest codons by ribosome residence time (RRT), purple arrows show the seven rarest codons by total usage, and <xref ref-type="fig" rid="fig2s1">Figure 2—figure supplement 1</xref> shows the correlation between codon usage and translation speed. Blue shows Start and alternative Start codons (ATG, TTG, ATT, ATA). Ratios above 1 show enrichment in the first 40 amino acids. Typically, the rarest codons are enriched. (<bold>B</bold>). Absolute usage of each leucine codon in the SIT. The absolute usage frequency of each leucine codon is shown globally, and for the first 40 amino acids. Rare codons are still rare in the SIT, just not as rare as elsewhere. The same pattern holds for the other amino acids.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-89656-fig2-v1.tif"/></fig><fig id="fig2s1" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 1.</label><caption><title>Codon speed and codon usage are correlated.</title><p>(<bold>A</bold>) Rare codons are translated slowly. Each dot represents a sense codon. The x-axis displays the translation speed of each codon (modified from <xref ref-type="bibr" rid="bib14">Gardin et al., 2014</xref>); the y-axis displays the global frequency of usage of each codon. The correlation is 0.64, p&lt;0.001. (<bold>B</bold>) The first 40 codons are enriched for rare codons. Each dot represents a sense codon. The relative usage of each type of codon in the first 40 codons of genes (i.e. in the slow initial translation , SIT) (y-axis) is displayed against global codon usage (x-axis). The correlation is –0.61, p&lt;0.001. (<bold>C</bold>) The first 40 codons are enriched for slow codons. Each dot represents a sense codon. The relative usage of each type of codon in the first 40 codons of genes (i.e. in the SIT) is displayed against codon translation speed (i.e. 1/ribosome residence time (RRT), <xref ref-type="bibr" rid="bib14">Gardin et al., 2014</xref>). The correlation is –0.45, p&lt;0.001.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-89656-fig2-figsupp1-v1.tif"/></fig></fig-group><p>For almost all amino acids, with isoleucine (ATC, ATA, ATT) being the only clear exception, we saw relative enrichment of the rarest and generally slowest codons. In particular, there were notable enrichments of the three rarest, slowest arginine codons (CGA, CGC, CGG) (likely because of their use in N-terminal signal sequences, see below) and the slow, rare codons for proline (CCC, CCG), leucine (CTC), glycine (GGG), and cysteine (TGC) (<xref ref-type="fig" rid="fig2">Figure 2A</xref>). These enrichments can explain most of the slow initial translation.</p><p>In addition, in the first 40 codons after but not including the initiator ATG, we saw notable depletion of the canonical Start codon ATG (by nearly 50%), and the alternative Start codons ATT and TTG (<xref ref-type="bibr" rid="bib11">Eisenberg et al., 2020</xref>; <xref ref-type="fig" rid="fig2">Figure 2A</xref>) by lesser but still significant amounts. These three codons are very fast (<xref ref-type="supplementary-material" rid="supp1">Supplementary file 1</xref>), and the depletion of these fast codons would make the average translation speed slower. Possibly Start codons are depleted to reduce the possibility of translation initiating at the wrong place. We recalculated translation speeds after assigning ATG a neutral speed (<xref ref-type="supplementary-material" rid="supp2">Supplementary file 2</xref>). This reduced the difference in translation speed between the SIT and the rest of the gene by about 15% of the difference. Since the three most commonly used alternative Start codons are ATT, TTG, and ATA (<xref ref-type="bibr" rid="bib11">Eisenberg et al., 2020</xref>), we also neutralized these by assigning them a neutral RRT (1.0189), in addition to neutralizing ATG. After neutralization of all four codons, the difference in translation speed between the SIT and the rest of the gene was reduced by about 40%, a significant change (<xref ref-type="supplementary-material" rid="supp2">Supplementary file 2</xref>). Even so, the remaining slow initial translation was highly significant. Thus, the depletion of Start codons contributes significantly to slow initial translation, but enrichment for rare, slow codons contributes even more.</p><p>(ATG and alternative Start codons are also depleted in the other two reading frames, but these depletions have indirect effects on the in-frame codons, such that there are roughly off-setting effects on translation speed. For instance, the depletion of TGG (Trp) (<xref ref-type="fig" rid="fig2">Figure 2</xref>) is likely partly due to depletion of xx<bold>A TG</bold>G, but since TGG is a slow codon, this depletion increases 5’ translation speed.)</p></sec><sec id="s2-3"><title>Rare codons are rare in the first 40 codons, just not as rare as elsewhere</title><p>Although the proportion of rare, slow codons in the SITs is relatively higher than in the body of genes, in absolute terms rare codons are still rare compared to more common synonymous codons (e.g. <xref ref-type="fig" rid="fig2">Figure 2B</xref>, Leu codons). That is, rare, slow codons are still strongly disfavored in the SIT, though they are less strongly disfavored than elsewhere. This was true for all rare codons.</p></sec><sec id="s2-4"><title>Why are there relatively more rare, slow codons at 5’ ends?</title><p>Our analysis is consistent with that of Tuller et al. to the extent that we find a slight relative enrichment of rare, slow codons near the 5’ ends of coding regions. Tuller et al. interpret the slow translation ramp as an adaptation—they believe there is a selection for slow codons near the 5’ end to enhance the efficiency of translation. But there are other possibilities.</p></sec><sec id="s2-5"><title>The Young Spandrel hypothesis</title><p>We noticed (see below) that the N-termini of yeast genes are often poorly conserved, and otherwise highly homologous genes often vary at the N-termini between different closely related species. This suggests a different idea: N-termini are unstable and variable in evolution. They form de novo from new DNA sequence, and so all codons may initially occur at similar frequencies. De novo formation of a N-terminus could occur from use of a new Start codon (<xref ref-type="bibr" rid="bib2">Bazykin and Kochetov, 2011</xref>; <xref ref-type="bibr" rid="bib21">Kochetov, 2008</xref>). Since these N-termini are, on average, younger than the remainder of the gene, selection has worked on them for a shorter time. Therefore, selection against rare, slow codons may be less complete, and N-termini, due to their relative youth, may still retain some extra rare codons.</p><p>In this idea, in contrast to Tuller et al. the slight excess of rare codons near N-termini is not adaptive; it is not at all a product of selection. Instead, in the words of <xref ref-type="bibr" rid="bib15">Gould and Lewontin, 1979</xref>, it is a spandrel. It is a non-adaptive by-product of something else, in this case, the evolutionary instability of N-termini. This idea is explored below.</p></sec><sec id="s2-6"><title>Poor 5’ conservation is a feature of many yeast genes</title><p>We picked example genes for illustration. We used protein-protein BLAST at NCBI to blast several query genes against species of the subphylum <italic>Saccharomycotina</italic> (but having subtracted out all of <italic>Saccharomyces</italic>). <xref ref-type="fig" rid="fig3">Figure 3</xref> shows that for these examples, the middle portions of the proteins are highly conserved, but the N-termini are not. We suggest that these non-conserved N-terminal regions are young in evolution, and therefore likely contain an excess of rare codons.</p><fig id="fig3" position="float"><label>Figure 3.</label><caption><title>The N-termini of proteins can vary in evolution.</title><p>BLAST of four example <italic>S. cerevisiae</italic> proteins against proteins in the subphylum ‘<italic>Saccharomycotina’</italic> (taxid: 147537) (excluding <italic>Saccharomyces,</italic> taxid 4930) was performed. Top hits are shown. Red regions indicate homology with an alignment score &gt;200, while white indicates no detected homology (BLAST default parameters). Even though all hits have high to moderate homology towards the center of the protein, many have little or no homology at the N-terminus.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-89656-fig3-v1.tif"/></fig><p>In the next section, we ask whether the findings of <xref ref-type="fig" rid="fig3">Figure 3</xref> can be generalized to proteins of <italic>S. cerevisiae</italic>.</p></sec><sec id="s2-7"><title>Method of scoring N-terminal conservation, and rationale for using <italic>Saccharomycotina</italic></title><p>We investigated N-terminal protein conservation using a quantitative approach. We ran local protein-protein BLAST for all <italic>S. cerevisiae</italic> genes against sequences of the <italic>Saccharomycotina</italic> subphylum, omitting <italic>Saccharomyces cerevisiae. Saccharomycotina</italic> was chosen because almost every gene from <italic>S. cerevisiae</italic> has a recognizable, conserved homolog in almost every species in <italic>Saccharomycotina</italic>, and yet the evolutionary distances are long enough that there is considerable sequence variability. We excluded species of <italic>Saccharomyces</italic> as they are too closely related to <italic>S. cerevisiae</italic>, and they are very numerous in sequence collections, and would overwhelm results from the other members of <italic>Saccharomycotina</italic>. However, we believe that this choice of subphylum does not greatly affect the final result.</p><p>Conservation at the N-terminus was calculated as the weighted proportion of yeast species with sequence matches (a match by default BLAST parameters) beginning in the first 40 amino acids. The lowest conservation score is 0 (no hits in the first 40 amino acids), whereas the highest conservation score is 40 indicating that every species had a match (default BLAST parameters) starting at the first amino acid (<xref ref-type="supplementary-material" rid="supp3 supp4">Supplementary file 3 and 4</xref>). The length of genes is negatively correlated with the conservation score, especially at the N-terminus (rho = –0.47; p &lt;0.001), but also for the rest of the gene (rho = −0.37, p&lt;0.001)—that is, short genes tend to be more conserved.</p></sec><sec id="s2-8"><title>N-termini are variable and poorly conserved</title><p>We measured protein conservation across more than 3000 <italic>S. cerevisiae</italic> proteins with orthologues among 822 closely related yeasts from <italic>Saccharomycotina</italic>. For each protein, we developed a conservation score for the first 40 amino acids to represent the N-termini, an equivalent conservation score for the middle 40 amino acids, and an equivalent conservation score for the C-terminal 40 amino acids (Methods and materials; <xref ref-type="supplementary-material" rid="supp3 supp4">Supplementary file 3 and 4</xref>).</p><p>Strikingly, the N-termini of <italic>S. cerevisiae</italic> orthologs had conservation scores that were much lower, and very differently distributed than the middle of the same orthologs (<xref ref-type="fig" rid="fig4">Figure 4</xref>). The first 40 amino acids had a flat distribution of protein conservation scores, indicating high levels of variability amongst these orthologs. That is, many of the orthologs had no detectable homology with the first 40 amino acids of the <italic>cerevisiae</italic> protein.</p><fig-group><fig id="fig4" position="float"><label>Figure 4.</label><caption><title>Conservation of <italic>S</italic>. <italic>cerevisiae</italic> proteins over the N-terminal, Middle, and C-terminal 40 amino acids.</title><p><italic>S. cerevisiae</italic> proteins were blasted against proteins of <italic>Saccharomycotina</italic> (excluding <italic>cerevisiae</italic>). ‘Conservation Scores’ (Methods and materials) were calculated for the N-terminal, Middle, and C-terminal 40 amino acids of the <italic>S. cerevisiae</italic> proteins. Scores range from 0 (no conservation) to 40 (perfect conservation). The frequency of each conservation score (3964 <italic>S</italic>. <italic>cerevisiae</italic> proteins) was plotted.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-89656-fig4-v1.tif"/></fig><fig id="fig4s1" position="float" specific-use="child-fig"><label>Figure 4—figure supplement 1.</label><caption><title>Comparison of conservation scores at the N- and C-termini.</title><p>Gray, N-terminal conservation scores. Red, C-terminal conservation scores.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-89656-fig4-figsupp1-v1.tif"/></fig></fig-group><p>In contrast, the middle 40 amino acids were highly conserved, with conservations scores peaking at 40, the highest possible score. It Is evident that for the middle 40 amino acids, a large fraction of orthologs had a region of high homology to the <italic>S. cerevisiae</italic> protein, whereas this was not true for the N-termini. Finally, the last 40 amino acids had conservation scores similar to those of the first 40 amino acids, though a bit higher (more conserved) see <xref ref-type="fig" rid="fig4s1">Figure 4—figure supplement 1</xref> for a comparison. These results suggest both ends of the gene ‘breathe,’ gaining and losing new sequences during evolution, whilst the middles stay constant. Thus the ends of genes are younger than their middles. At their first formation, they would likely have contained some rare codons, which selection may not yet have had time to remove.</p></sec><sec id="s2-9"><title>The 3’ ends of genes also have slightly slow translation</title><p>As shown in <xref ref-type="fig" rid="fig4">Figure 4C</xref>, the C-termini of proteins have poor conservation, like the N-termini. Therefore, the Spandrel hypothesis predicts slow translation at 3’ ends. We calculated translation speeds at 3’ ends, and again found slightly slow translation (<xref ref-type="fig" rid="fig5">Figure 5</xref>). This was not statistically significant over the last 40 codons, but it was significant over the last 100 codons (<xref ref-type="fig" rid="fig5">Figure 5</xref>) and the last 120 codons (<xref ref-type="supplementary-material" rid="supp2">Supplementary file 2</xref>). Like the 5’ end, there was a slightly increased relative frequency of rare codons (<xref ref-type="fig" rid="fig5s1">Figure 5—figure supplement 1</xref>), but unlike the 5’ end, ATG was not depleted (<xref ref-type="fig" rid="fig5s1">Figure 5—figure supplement 1</xref>).</p><fig-group><fig id="fig5" position="float"><label>Figure 5.</label><caption><title>Translation speed at 3’ ends.</title><p>Translation speeds at the 3’ ends of genes were calculated using ribosome residence time (RRT) (<xref ref-type="bibr" rid="bib14">Gardin et al., 2014</xref>; <xref ref-type="supplementary-material" rid="supp1">Supplementary file 1</xref> for RRT values). The average speed over the last 40 amino acids is about 0.1% slower than in the rest of the gene, not statistically significant. The average speed over the last 100 amino acids is about 0.19% slower, which is significantly different (p=0.028).</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-89656-fig5-v1.tif"/></fig><fig id="fig5s1" position="float" specific-use="child-fig"><label>Figure 5—figure supplement 1.</label><caption><title>Codon characteristics at the beginnings and ends of yeast genes.</title><p>Red arrows identify the seven slowest codons (CCG, CGA, CGG, GGG, CGC, CCC, and TGG), purple arrows identify the seven rarest codons (CGG, CGC, CGA, TGC, CCG, CTC, and GGG), and blue arrows identify four Start codons (ATA, ATG, ATT, and TTG). Codons for each amino acid are arranged from least to most used, left to right.(<bold>A</bold>) Each bar is a ratio of the codon usage among the last 125 codons relative to codon usage for the remainder of genes. (<bold>B</bold>) Codon usage among the first 40 codons relative to the last 125 codons.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-89656-fig5-figsupp1-v1.tif"/></fig><fig id="fig5s2" position="float" specific-use="child-fig"><label>Figure 5—figure supplement 2.</label><caption><title>Mitochondrial and ER signal sequences.</title><p>467 mitochondrial genes were defined according to Williams et al. (14), and 222 proteins with a predicted ER signal peptide were defined according to Jan et al. (15). (<bold>A</bold>) The relative initial translation speeds (SITs) of mitochondrial, ER, and all other proteins were characterized. (<bold>B</bold>) The N-terminal protein conservation scores were characterized for the three sets of proteins.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-89656-fig5-figsupp2-v1.tif"/></fig></fig-group><p>There may be at least three reasons why translation at 3’ ends is not as slow as at 5’ ends. First, at 3’ ends, Start codons and alternative-Start codons (which are fast) are not depleted (because at 3’ ends, there is no issue of generating incorrect translation initiation sites), and this retention of fast codons tends to make 3’ ends faster than 5’ ends.</p><p>Second, there are two sets of genes, mitochondrial genes, and ER (Endoplasmic Reticulum) genes, that have especially slow 5’ translation. Tuller et al. characterized several functional groups of genes for the amplitude and length of their slow initial translation ramp. The group with the greatest amplitude was the group of 282 genes for ‘Mitochondrial organization.’ This group is dominated by genes for proteins imported into mitochondria. We believe this especially slow translation is due to N-terminal signal sequences. Mitochondrial import depends on an N-terminal signal sequence. A typical mitochondrial signal sequence has an average of 25 residues, is highly enriched for arginine, and has relatively little sequence conservation (e.g. cluster I of <xref ref-type="bibr" rid="bib12">Fukasawa et al., 2015</xref>). Since four of the 10 slowest codons are for arginine, and since little sequence conservation is required in a mitochondrial signal sequence, these signal sequences seem good candidates for regions that could vary rapidly in evolution, and have slow initial translation (thanks largely to rare, slow Arg codons). Indeed, we found that for a set of 467 mitochondrial proteins (<xref ref-type="bibr" rid="bib37">Williams et al., 2014</xref>) initial translation was about 2.3% slower than in the rest of the genes, versus only 1.04% slower for genes with no mitochondrial or ER signal sequence (<xref ref-type="supplementary-material" rid="supp2">Supplementary file 2</xref> and <xref ref-type="fig" rid="fig5s2">Figure 5—figure supplement 2</xref>). We had similar findings, but to a lesser extent, for proteins with an ER signal sequence, which is also rich in basic residues (<xref ref-type="fig" rid="fig5s2">Figure 5—figure supplement 2</xref>). In this case, <xref ref-type="bibr" rid="bib26">Pechmann et al., 2014</xref> have argued that a cluster of rare, slow codons 35–40 codons from the N-terminus provide a translational pause that allows the Signal Recognition Particle time to recognize the signal sequence. Both kinds of explanations could be true.</p><p>Third, as an adaptation argument, 5’ ends could sometimes be selected for poor translation to produce an appropriately small amount of protein, and this would sometimes favor rare codons (see below).</p><p>(<xref ref-type="bibr" rid="bib7">Cope et al., 2018</xref>) had similar findings for N-terminal signal peptides of <italic>E. coli</italic>, which are enriched in translationally inefficient codons. Like us, they suggested selection for codon usage was relatively weak (or evolutionarily brief) at 5’ ends, and they cited the ‘Spandrel’ idea of <xref ref-type="bibr" rid="bib15">Gould and Lewontin, 1979</xref>: that is, the inefficient codons might have arisen for a non-adaptive reason, and persisted because of weak (or brief) selection.</p></sec><sec id="s2-10"><title>5’ translation speeds positively correlate with 5’ conservation scores</title><p>If the ‘Young Spandrel’ hypothesis is true, and slow 5’ translation is partly caused by evolutionary instability of 5’ ends, then there should be a correlation between encoded slow translation, and poor N-terminal conservation. Our model predicts the least conserved N-termini to have the slowest translation (i.e. rarest codons), and, <italic>vice versa</italic>, the termini with the slowest translation should have the lowest conservation. To test this, we ranked all genes by the conservation scores of their first 40 amino acids. We then divided this ranked list into thirds. For each of the thirds (i.e. the bottom, middle, and top conservation scores) we plotted the average relative initial translation speeds.</p><p>As shown in <xref ref-type="fig" rid="fig6">Figure 6A</xref>, the genes with the most poorly conserved N-termini also had the slowest initial translation, while the genes with the most conserved N-termini had the fastest initial translation, supporting the Spandrel hypothesis, and opposite to the Ramp hypothesis.</p><fig-group><fig id="fig6" position="float"><label>Figure 6.</label><caption><title>Slow initial translation is correlated with poor N-terminal conservation.</title><p>(<bold>A</bold>) Proteins were grouped by their N-terminal conservation scores (top, middle, and bottom thirds), and then the relative initial translation rate was plotted for each group. More conserved N-termini have a faster initial translation. (<bold>B</bold>) Proteins were grouped by their initial translation rate (Slow, SIT; Medium, MIT, or Fast, FIT), and then the N-terminal conservation scores were plotted for each group. Genes with faster initial translation have more conserved N-termini. Relative Initial Translation Speed is the log2 of (average ribosome residence time, RRT of the first 40 amino acids divided by the average RRT of the rest of the same gene) (Methods and materials).</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-89656-fig6-v1.tif"/></fig><fig id="fig6s1" position="float" specific-use="child-fig"><label>Figure 6—figure supplement 1.</label><caption><title>Slow 3’ translation is correlated with poor C-terminal conservation.</title><p>(<bold>A</bold>) Proteins were grouped by their C-terminal conservation scores (top, middle, and bottom thirds), and then the terminal translation rate was plotted for each group. More conserved C-termini have faster terminal translation. (<bold>B</bold>) Proteins were grouped by their C-terminal translation rate (top, middle, and bottom thirds), and then the C-terminal conservation scores were plotted for each group. Genes with faster terminal translation have more conserved C-termini.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-89656-fig6-figsupp1-v1.tif"/></fig></fig-group><p>We also looked at the correlation in the other direction (<xref ref-type="fig" rid="fig6">Figure 6B</xref>). We ranked genes by their relative initial translation speed, and divided the ranked list into thirds, then plotted N-terminal conservation scores. Again the effects are correlated: genes with the slowest initial translation have the lowest N-terminal conservation scores. Thus, overall, there is a strong correlation between N-terminal instability (i.e. newness in evolution, low conservation scores) and slow initial translation (i.e. the presence of slow/rare codons).</p><p>These correlations (i.e. between poor conservation and slow translation; and between slow translation and poor conservation) were also seen at the 3’ ends of genes (<xref ref-type="fig" rid="fig6s1">Figure 6—figure supplement 1</xref>).</p></sec><sec id="s2-11"><title>The Ramp hypothesis is inconsistent with observations of ribosome density and gene expression</title><p>In the Tuller ‘Ramp’ hypothesis, in which the purpose of the slow translational ramp is to queue ribosomes and prevent collisions, genes with the highest ribosome occupancy would be in the most danger of ribosome collisions, and would, therefore, presumably have pronounced SITs. SITs might not be necessary on genes with low ribosome density, since there would not be much danger of collision in any case. To test this, we used information from <xref ref-type="bibr" rid="bib1">Arava et al., 2003</xref>, which measured the density of ribosomes on all <italic>S. cerevisiae</italic> mRNAs (<xref ref-type="bibr" rid="bib1">Arava et al., 2003</xref>; <xref ref-type="fig" rid="fig7">Figure 7</xref>). We ranked genes by ribosome density, then grouped them in thirds. Opposite to the expectation from the Tuller et al. ‘Ramp’ theory, the genes with the highest ribosome densities had the fastest initial translation, whereas the genes with the lowest ribosome densities had the slowest initial translation (<xref ref-type="fig" rid="fig7">Figure 7C</xref>). While these findings are opposite to the expectation of the ‘Ramp’ theory, they are consistent with the spandrel theory, because genes with high ribosome density would be subject to more intense selection against slow codons, thus leading to faster 5’ ends. Furthermore, analysis of Conservation Scores on the same genes showed that the genes with the lowest ribosome densities also had the lowest Conservation Scores (<xref ref-type="fig" rid="fig7">Figure 7D</xref>), as predicted by the Spandrel hypothesis.</p><fig id="fig7" position="float"><label>Figure 7.</label><caption><title>Genes with high levels of expression, and high ribosome densities, generally have rapidly-translated N-termini, and high N-terminal conservation scores.</title><p>(<bold>A</bold> and <bold>B</bold>) Genes were grouped by expression level (bottom, middle, and top)(except that genes with fewer than 10 read-counts were omitted to reduce noise) (<xref ref-type="bibr" rid="bib24">Lipson et al., 2009</xref>). In A, the initial translation rate is shown; in B, the conservation scores are shown. The correlation between speed and transcript abundance fails for the bottom third of genes; possibly these are genes expressed at high levels under other conditions (e.g. meiosis and sporulation). (<bold>C</bold> and <bold>D</bold>) Genes were grouped by ribosome density (<xref ref-type="bibr" rid="bib1">Arava et al., 2003</xref>) as a measure of intensity of translation. In C, the initial translation rate is shown; in D, the conservation scores are shown. High ribosome density correlates with high initial translation speed and high conservation score.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-89656-fig7-v1.tif"/></fig><p>Similarly, in the ‘Ramp’ hypothesis, ribosome collisions on highly-expressed genes would presumably have more serious consequences for the cell than collisions on poorly-expressed genes, and so highly-expressed genes ought to have the most pronounced SITs. To test this, we used transcriptomic information from <xref ref-type="bibr" rid="bib24">Lipson et al., 2009</xref>, which measured the number of mRNA transcripts for <italic>S. cerevisiae</italic> genes (<xref ref-type="bibr" rid="bib24">Lipson et al., 2009</xref>). Again, we ranked genes by expression, then grouped them by thirds. Exactly contrary to the ‘Ramp’ hypothesis, we found that genes with the highest expression had the fastest initial translation (<xref ref-type="fig" rid="fig7">Figure 7A</xref>). In this analysis, the middle and bottom genes are not significantly different from each other; possibly some of the poorly expressed genes are inducible genes that would be highly expressed under some other condition (e.g. the sporulation genes, the <italic>GAL</italic> genes). In addition, the most highly expressed genes had the highest conservation scores, consistent with the Spandrel hypothesis (<xref ref-type="fig" rid="fig7">Figure 7B</xref>).</p></sec><sec id="s2-12"><title>Experimentally, encoded slow initial translation does not increase gene expression; the opposite is true</title><p>Tuller et al. hypothesized that slow initial translation was adaptive, and improved the efficiency of translation and gene expression by minimizing ribosome collisions. However, in the Spandrel hypothesis, slow initial translation is not generated by selection and is not adaptive. It might not have any effect on gene expression, but, if anything, slow translation might reduce gene expression. While informatics is wonderful, it is always nice to do an experiment, and in this section, we present direct experimental results regarding the effect of encoded slow, medium, and fast initial translation on gene expression.</p><p>We used a gene expression reporter based on EKD1024 (<xref ref-type="bibr" rid="bib5">Brule et al., 2016</xref>) (Methods and materials). In this construct, GFP is the reporter, but for accuracy it is normalized against a divergently transcribed red fluorescent protein (<xref ref-type="fig" rid="fig8s1">Figure 8—figure supplement 1</xref>). Thus, GFP expression is reported as a GFP/RFP ratio. Although the reporter is GFP, the N-terminal region of this particular protein is derived from yeast <italic>HIS3</italic>, not GFP, and likely has little if any effect on the fluorescence of the GFP fused downstream (<xref ref-type="bibr" rid="bib10">Dean and Grayhack, 2012</xref>; <xref ref-type="bibr" rid="bib13">Gamble et al., 2016</xref>; <xref ref-type="bibr" rid="bib27">Pédelacq et al., 2006</xref>). We used synonymous slow, medium, or fast codons to recode some of the codons in the first 41 amino acids of this GFP reporter to generate three reporters with slow, medium, or fast translation over the first 41 amino acids. We emphasize that the amino acid sequences of the three constructs were identical. The slow, medium, and fast average RRT values over the first 41 codons were 1.20, 1.04, and 0.93, respectively. That is, this SIT is slower than most natural SITs, and this FIT is faster than most natural FITs, but the difference is moderate.</p><p>As shown in <xref ref-type="fig" rid="fig8">Figure 8</xref> (left three bars), the SIT did not improve expression of GFP, contrary to Tuller et al. In fact, the GFP with the SIT was expressed at only 71% of the level of the GFP with the FIT. It was surprising to us that the difference was this large—again, recoding was limited to codons within the first 41, and the protein sequences were identical.</p><fig-group><fig id="fig8" position="float"><label>Figure 8.</label><caption><title>Slow initial translation inhibits gene expression.</title><p>Left three bars. A synthetic GFP was constructed with a leader amino acid sequence that had little effect on GFP. The leader sequence was recoded to give slow (SIT), medium (MIT), or fast (FIT) translation speed over the first 41 amino acids, without changing the amino acid sequence—i.e., the SIT, MIT, and FIT had identical amino acid sequences, but different average ribosome residence times (RRTs). Each construct (SIT, MIT, FIT) was integrated in a single copy at the <italic>ADE2</italic> locus, and 25 independently-transformed strains were picked, and GFP fluorescence was measured for each, and the RFP-normalized mean was plotted. Numerical values were: SIT, 1.66; MIT, 1.80, FIT, 2.29. GFP was normalized to RFP expressed from the same reporter molecule, but RFP fluorescence hardly changed amongst the transformants, and non-normalized GFP would have given very similar results. Slower initial translation reduced gene expression. Right three bars. As above, a Putative ribosome collision site (PCS) (CGA-CGG) was inserted between the leader and the GFP. Again, slower initial translation reduced gene expression. Values were: SIT:PCS, 0.69, MIT:PCS, 0.74, FIT:PCS, 0.99.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-89656-fig8-v1.tif"/></fig><fig id="fig8s1" position="float" specific-use="child-fig"><label>Figure 8—figure supplement 1.</label><caption><title>Structure of the GFP reporters.</title><p>These reporters were adapted from <xref ref-type="bibr" rid="bib5">Brule et al., 2016</xref>; <xref ref-type="bibr" rid="bib13">Gamble et al., 2016</xref> . A leader sequencer (purple), originally from <italic>HIS3</italic>, is appended upstream of GFP. (<bold>A</bold>) For the first three constructs, recoding of some residues within the first 41 codons with synonymous codons gave either a slow (SIT), medium (MIT), or fast (FIT) initial translation rate. Protein sequences were preserved. (<bold>B</bold>) Three analogous reporters were made with a putative ribosome collision site (PCS) at codon positions 68 and 69 (still upstream of GFP sequences). The PCS was the codon pair CGA-CGG, two rare Arg codons.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-89656-fig8-figsupp1-v1.tif"/></fig></fig-group><p>Another possibility is that a SIT can protect against ribosome collisions when there is a site downstream that induces ribosome collisions. Sites thought to induce ribosome collisions include rare Arg-Arg codon pairs (<xref ref-type="bibr" rid="bib9">Dao Duc and Song, 2018</xref>; <xref ref-type="bibr" rid="bib31">Tesina et al., 2020</xref>). We, therefore, introduced the codon pair CGA-CGG (replacing Asn-Asp, AAT-GAT) downstream of the first 41 amino acids, but still upstream of important GFP residues. Indeed, this single CGA-CGG codon pair, potentially inciting ribosome collisions, caused a large reduction--about 50%--in the expression of GFP (<xref ref-type="fig" rid="fig8">Figure 8</xref>, right three bars). The reduction was about the same in the SIT, MIT, and FIT constructs. In this case, with putative collision sites, the GFP with the SIT was expressed at only 67% of the level of the equivalent GFP with the FIT. That is, this SIT (a fairly extreme SIT) did not at all protect against the putative ribosome collisions—if anything, it made things slightly worse. This result suggests there is no benefit to ‘queuing’ ribosomes, if queuing even occurs. Instead, the fastest-translating gene once again gave the highest expression, and the highest relative expression, despite the collision site.</p></sec></sec><sec id="s3" sec-type="discussion"><title>Discussion</title><p>Tuller et al. found that the 5’ ends of genes are translated slowly because of the codons used at 5’ ends, and posited that this was a selective advantage because it somehow increased the efficiency of translation. However, this theory predicts positive correlations between slow initial translation and high gene expression, and slow initial translation and high overall (that is, on the whole gene) ribosome density. In fact, by informatic analysis of existing data, we find the correlations are opposite to those predicted by the ‘Ramp’ model. Most importantly, an experiment in which codon usage at 5’ ends was changed shows that faster 5’ codons cause higher gene expression, exactly opposite to the prediction of the ‘Ramp’ hypothesis. We believe no ramp is needed.</p><p>Tuller et al. showed that a region of slow translation is encoded, using slowly-translated codons, and it is specifically this idea of encoded slow translation that we are addressing. This encoded slow translation is a small effect—translation is slowed by 1% to 3%. However, in addition to ‘encoded’ slow translation, there is evidence for slow translation at 5’ ends by other, unknown mechanisms, with apparently much larger amplitudes, perhaps greater than 50%. Ribosome profiling experiments show an increased density of ribosome footprints near the 5’ end, independent of encoding (<xref ref-type="bibr" rid="bib36">Weinberg et al., 2016</xref>), which could be due to slow translation. It is now known that the very high 5’ density of footprints in early ribosome profiling studies was due to the use of cycloheximide as a first step to stop translation. Addition of cycloheximide to growing cells allowed ribosomes to initiate at Start codons, but did not allow elongation, hence there was a pile-up of ribosomes near the 5’ end. More recently, flash-freezing, rather than cycloheximide, has been used as the first step in stopping translation. However, even in these studies, there is about a 50% increase in ribosome density near 5’ ends (<xref ref-type="bibr" rid="bib36">Weinberg et al., 2016</xref>). And yet, even in these flash-freezing protocols, cycloheximide is still used at a later step to prevent elongation when extracts are thawed, and this cycloheximide usage could again result in an artifactual increase in ribosome density at 5’ ends. Alternatively, the increased density of ribosomes at 5’ ends could mean that some proportion of ribosomes fall off the mRNA as they progress (<xref ref-type="bibr" rid="bib36">Weinberg et al., 2016</xref>). Consistent with the latter idea, ribosome profiling studies show a general trend towards lower ribosome densities at more 3’ positions in translating mRNAs (Weinberg et al., their Figure S7), and studies using other experimental approaches have shown a general decrease in ribosome number or density as one progresses along a gene (<xref ref-type="bibr" rid="bib3">Bonderoff and Lloyd, 2010</xref>; <xref ref-type="bibr" rid="bib34">Verma et al., 2019</xref>). A different idea was proposed by <xref ref-type="bibr" rid="bib29">Shah et al., 2013</xref> in a theory paper, which suggested this apparent slow translation could be an informatic artifact caused by rapid translational initiation (and, therefore, high ribosome density) on short genes. But none of these ideas addresses the fact found by Tuller that the 5’ ends of genes are enriched in rare, slow codons.</p><p>We considered that an increased density of ribosomes at the 5’ end could be because some genes have additional ATG Start codons, sometimes upstream and sometimes downstream of the annotated Start, and translation of short open reading frames from these additional Start codons could contribute to ribosome density at the 5’ end. Using the program ‘Frameshift Detector’ (<xref ref-type="bibr" rid="bib38">Yurovsky et al., 2022</xref>) and ribosome profiling data, we quantitated the fraction of out-of-frame ribosomes both globally, and within the first 150 nucleotides of genes. We found the global proportion of out-of-frame ribosomes is about 13%, and the proportion of out-of-frame ribosomes in the first 150 nucleotides is about 14.5%. Although this increased 5’ out-of-frame ribosome presence of about 1.5% is highly significant (p~10<sup>–28</sup>), it is not nearly big enough to explain the observed 5’ increase in ribosome density (<xref ref-type="bibr" rid="bib36">Weinberg et al., 2016</xref>, their Figure 1C).</p><p>In any case, by direct experiment, we find that encoding a slower 5’ end using slow synonymous codons reduces gene expression. In particular, even when a ribosome collision site was placed downstream, the effect of the collision site was not at all ameliorated by encoded slow translation upstream of the collision site. This seems strong evidence against the idea that slow initial translation is a defense against collisions. The basis of the idea that slow initial translation could possibly be a defense against collisions is not clear to us. Regions of slow translation would not affect the gaps between ribosomes, if measured as times, and especially not if measured at a constant finish line, such as a putative collision site.</p><p>An issue in the GFP reporter experiment is that the mRNA sequences are necessarily different, and so there are different mRNA structures. We achieved fast and slow sequences by recoding with synonymous codons, so amino acids are identical, so there are no issues of, e.g., co-translational protein folding, or amino acid interaction with the ribosome exit tunnel. But of course, the RNA structures are at least slightly different, and RNA structures at the 5’ end are known to affect translation initiation (<xref ref-type="bibr" rid="bib36">Weinberg et al., 2016</xref>; <xref ref-type="bibr" rid="bib6">Burkhardt et al., 2017</xref>; <xref ref-type="bibr" rid="bib8">Cuperus et al., 2017</xref>; <xref ref-type="bibr" rid="bib17">Gu et al., 2010</xref>; <xref ref-type="bibr" rid="bib18">Hall et al., 1982</xref>; <xref ref-type="bibr" rid="bib22">Kudla et al., 2009</xref>; <xref ref-type="bibr" rid="bib25">Nackley et al., 2006</xref>). Generally, more open mRNA structures are more favorable both for translation initiation and for translation speed. To fully disentangle these effects is difficult. But whether the increased gene expression we see for the fast encoding is being generated mainly by fast translation, or by efficient initiation, in neither case is there an argument that slow translation is efficient, or that it protects against collisions.</p><p>The observation that 5’ ends have low conservation, likely because of instability in evolution, provides a completely different explanation for the enrichment of slow codons at 5’ ends. In this ‘Spandrel’ hypothesis, N-termini frequently change in evolution, gathering new 5’ sequences de novo. These would contain all codons at similar frequencies—i.e., ‘rare’ codons would not be especially rare. Although rare codons would eventually be removed by selection, the fact that N-termini are relatively young means that this process might not be complete for all genes, and so some rare, slow codons still remain. These explain the initial region of encoded slow translation. This hypothesis is highly consistent with the observed correlations between slow initial translation and low gene expression; and slow initial translation and low ribosome density, and with the results of gene expression experiments. It is also consistent with the region of slightly slow translation we observe at 3’ ends.</p><p>We have looked at the conservation of N-termini only in <italic>S. cerevisiae</italic>. However, Tuller et al. found that there is a region of encoded slow initial translation in genes of a wide variety of eukaryotes. We speculate that in these other cases, too, slow initial translation is a spandrel partly due to depletion of fast Start and alternative Start codons, and partly deriving from the turnover of 5’ ends. This in turn has implications for protein structure and evolution; for the interpretation of evolutionary sequence clocks; and for the rates of selection against rare codons. We note that <xref ref-type="bibr" rid="bib4">Bricout et al., 2023</xref> have also recently found that N- and C-termini of proteins evolve faster than the middles.</p><p>It was surprising to us that recoding just the first 41 codons of the GFP fusion protein from slow to fast increased the level of GFP expression by so much—about 30%. These 41 codons were originally derived from the yeast <italic>HIS3</italic> gene, and this increase in expression is roughly the proportional increase expected based on fully recoding the <italic>HIS3</italic> gene to preferred codons (<xref ref-type="bibr" rid="bib28">Presnyak et al., 2015</xref>). Because this recoding from slow to fast tends to replace G/C-rich codons with A/T-rich codons, recoding from slow to fast may decrease the stability of RNA structures near the 5’ end, and increase the accessibility of the cap. Decreased stability of mRNA structures could be responsible for the increase in gene expression, consistent with studies in both yeast (<xref ref-type="bibr" rid="bib36">Weinberg et al., 2016</xref>; <xref ref-type="bibr" rid="bib8">Cuperus et al., 2017</xref>) and <italic>E. coli</italic> (<xref ref-type="bibr" rid="bib22">Kudla et al., 2009</xref>).</p><p>Finally, again, in ‘The Spandrels of San Marco...,’ (<xref ref-type="bibr" rid="bib15">Gould and Lewontin, 1979</xref>) warned that not all biological phenomena are adaptive, and it is a mistake to assume that any particular characteristic of an organism must necessarily have been generated by natural selection. We believe the encoded slow initial translation of eukaryotic genes may be an example of this.</p></sec><sec id="s4" sec-type="materials|methods"><title>Materials and methods</title><sec id="s4-1"><title>Statistical calculations of relative initial rate of translation</title><p>Bioinformatics were performed on protein-coding open reading frames (ORFs) of <italic>Saccharomyces cerevisiae</italic> downloaded from the <italic>Saccharomyces</italic> Genome Database (SGD) website, as last modified on April 22, 2021. Protein-coding ORFs annotated as dubious or pseudogenes were not included in analyses. All statistics were performed using The R Project for Statistical Computing. Translation speed was measured using the ribosome residence time (RRT) which is a metric of the occupancy of ribosomes on each sense codon within the A-site (<xref ref-type="bibr" rid="bib14">Gardin et al., 2014</xref>). The RRT values we used are shown in <xref ref-type="supplementary-material" rid="supp1">Supplementary file 1</xref>; these are modified from the original RRT results of Gardin et al. by inclusion of the ribosome profiling data of <xref ref-type="bibr" rid="bib20">Jan et al., 2014</xref>.</p></sec><sec id="s4-2"><title>Relative initial translation speed</title><p>(<xref ref-type="bibr" rid="bib32">Tuller et al., 2010</xref>) focused on a ‘ramp’ of translation speed, where the first part of the gene has slow translation relative to the rest of the gene. The ramp thus refers to a rate. To quantitate this slow relative ramp for each gene, we calculated the average RRT for an initial window of the gene (e.g. 40 amino acids, see below), then divided by the average RRT of the rest of the gene. Thus, genes with a ‘slow ramp’ have a ratio of less than 1. We then took log2 of this ratio; genes with a slow ramp yield a negative number, and the more negative the number, the steeper the ramp.</p><p>For each gene, the relative initial translation speed (RIT) (explained above) was calculated across windows of the first 30, 40, 50... and 100 codons, with all windows being statistically significant for slow translation. For these RIT calculations, the first (start) codon was omitted since all protein-coding genes in this dataset except Q0075 start with ATG, and ATG is one of the fastest codons, which would skew the RIT. Similarly, the last (stop) codon was omitted.<disp-formula id="equ1"><mml:math id="m1"><mml:mrow><mml:mi mathvariant="normal">R</mml:mi><mml:mi mathvariant="normal">I</mml:mi><mml:mi mathvariant="normal">T</mml:mi><mml:mo>=</mml:mo><mml:mi mathvariant="normal">l</mml:mi><mml:mi mathvariant="normal">o</mml:mi><mml:mi mathvariant="normal">g</mml:mi><mml:mn>2</mml:mn><mml:mrow><mml:mo>(</mml:mo><mml:mfrac><mml:mrow><mml:mi mathvariant="normal">m</mml:mi><mml:mi mathvariant="normal">e</mml:mi><mml:mi mathvariant="normal">a</mml:mi><mml:mi mathvariant="normal">n</mml:mi><mml:mi mathvariant="normal">R</mml:mi><mml:mi mathvariant="normal">R</mml:mi><mml:mi mathvariant="normal">T</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi mathvariant="normal">c</mml:mi><mml:mi mathvariant="normal">o</mml:mi><mml:mi mathvariant="normal">d</mml:mi><mml:mi mathvariant="normal">o</mml:mi><mml:mi mathvariant="normal">n</mml:mi><mml:mi mathvariant="normal">s</mml:mi><mml:mn>2</mml:mn><mml:mo>:</mml:mo><mml:mi mathvariant="normal">w</mml:mi><mml:mi mathvariant="normal">i</mml:mi><mml:mi mathvariant="normal">n</mml:mi><mml:mi mathvariant="normal">d</mml:mi><mml:mi mathvariant="normal">o</mml:mi><mml:mi mathvariant="normal">w</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi mathvariant="normal">m</mml:mi><mml:mi mathvariant="normal">e</mml:mi><mml:mi mathvariant="normal">a</mml:mi><mml:mi mathvariant="normal">n</mml:mi><mml:mi mathvariant="normal">R</mml:mi><mml:mi mathvariant="normal">R</mml:mi><mml:mi mathvariant="normal">T</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi mathvariant="normal">c</mml:mi><mml:mi mathvariant="normal">o</mml:mi><mml:mi mathvariant="normal">d</mml:mi><mml:mi mathvariant="normal">o</mml:mi><mml:mi mathvariant="normal">n</mml:mi><mml:mi mathvariant="normal">s</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi mathvariant="normal">w</mml:mi><mml:mi mathvariant="normal">i</mml:mi><mml:mi mathvariant="normal">n</mml:mi><mml:mi mathvariant="normal">d</mml:mi><mml:mi mathvariant="normal">o</mml:mi><mml:mi mathvariant="normal">w</mml:mi><mml:mo>+</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>:</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi mathvariant="normal">l</mml:mi><mml:mi mathvariant="normal">a</mml:mi><mml:mi mathvariant="normal">s</mml:mi><mml:mi mathvariant="normal">t</mml:mi><mml:mtext> </mml:mtext><mml:mi mathvariant="normal">c</mml:mi><mml:mi mathvariant="normal">o</mml:mi><mml:mi mathvariant="normal">d</mml:mi><mml:mi mathvariant="normal">o</mml:mi><mml:mi mathvariant="normal">n</mml:mi><mml:mo>−</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:mfrac><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:math></disp-formula></p><p>Windows of the first 30, 40, 50... and 100 codons each had statistically significant depletion in translation speed compared to the body of genes. We chose to focus on the first 40 codons. All genes shorter than 303 nucleotides (translated into 100 amino acids) were omitted from all analyses. For all RIT analyses, 328 out of 6022 ORFs were omitted leaving a dataset of 5694 ORFs.</p></sec><sec id="s4-3"><title>Data</title><p>mRNA transcript readings, which we used as a proxy for gene expression, was acquired from <xref ref-type="bibr" rid="bib24">Lipson et al., 2009</xref>. Genes with a read count of less than 10 were omitted due to concerns about noise. Ribosome density measurements were acquired from <xref ref-type="bibr" rid="bib1">Arava et al., 2003</xref>. These values were calculated as the number of ribosomes, detected on an mRNA, divided by the nucleotide length of the gene (including the stop codon).</p></sec><sec id="s4-4"><title>Protein BLAST setup and diagnostics</title><p><italic>S. cerevisiae</italic> proteins were downloaded from the SGD (last modified on April 22, 2021). Proteins derived from ORFs annotated as dubious or pseudogene were omitted from analyses. The <italic>Saccharomycotina</italic> (Taxonomy ID: 147537) protein sequences were downloaded from NCBI using the links <ext-link ext-link-type="uri" xlink:href="https://www.ncbi.nlm.nih.gov/Taxonomy/Browser/wwwtax.cgi?mode=Info&amp;id=147537&amp;lvl=3&amp;lin=f&amp;keep=1&amp;srchmode=1&amp;unlock(taxonomy_id:147537)&amp;sort%20=%20organism_name">here</ext-link>.</p><p>We downloaded and compiled the source databases from DDBJ, EMBL, Genbank, RefSeq, PIR, and UniProtKB. Duplicate sequences were deleted. To perform local BLAST, we downloaded the NCBI BLAST software (version 2.13.0+) and used RStudio as a wrapper to operate the software; all default BLAST parameters were selected, except that the number of alignments was changed to the maximum value of 1000000000. Local protein BLAST of every <italic>S. cerevisiae</italic> protein (5694 proteins, see above) was performed against all genomes of the subphylum <italic>Saccharomycotina</italic>, but omitting all species in the genus <italic>Saccharomyces</italic> (net, 822 genomes). We eliminated submissions of duplicate species by limiting our database to the highest bit-scores from sequence hits derived from each unique species. We were only interested in sequences that had high homology with queried <italic>S. cerevisiae</italic> proteins, so all hits with bit-scores lower than 50 were omitted.</p><p>We wanted to compare conservation at the beginning of proteins with conservation at the middle and end of those same proteins. For this purpose, we split each <italic>S. cerevisiae</italic> protein into two halves (start to middle; middle to end), then blasted each half against all genomes in the subphylum <italic>Saccharomycotina</italic> (omitting <italic>Saccharomyces</italic>). We then calculated a ‘conservation score’ (see below) for the first 40 amino acids of the protein, and, identically, for the first 40 amino acids of the second half of the protein. (We describe the first 40 amino acids of the second half of the protein as the ‘middle,’ but in fact, the region is displaced 20 amino acids C-terminal from the exact middle.) In a parallel way, a conservation score is calculated for the last 40 amino acids of each protein. For example, the length of Swi5 is 709 amino acids, and therefore BLASTs of the first half spanned from 1:354, and BLASTs for the second half spanned from 355:709. Conservation scores were calculated for residues 1:40 (beginning) (from the BLASTs of the first half of the protein), and 355:394 (middle) and 670:709 (end) (from BLASTs of the second half of the protein). In total, BLAST of the first half of all queried <italic>S. cerevisiae</italic> proteins yielded a total of 477,749 high homology (minimum of bit-score of 50) sequence matches across 816 unique <italic>Saccharomycotina</italic> species, whereas BLAST of the second half of each protein yielded a total of 487,022 high homology matches across 812 unique <italic>Saccharomycotina</italic> species.</p><p>With respect to the above procedure, we note that we are relying on the BLAST algorithm to find regions of homology. Homology would be somewhat more easily found in the middle of sequences than at the ends because of seeding issues. It is for this reason that we divided proteins in half, and used a BLAST with the second half of the protein to find homologies with the first 40 amino acids of the second half. That is, in this procedure, for the middle homologies, the algorithm is being asked to find homologies at the end of a sequence, exactly as is the case for the first 40 and last 40 amino acids. We also used the alternative approach of finding homologies in the last 40 amino acids of the first half of the protein, with essentially identical results (<xref ref-type="fig" rid="fig4s1">Figure 4—figure supplement 1</xref>).</p></sec><sec id="s4-5"><title>Calculations of protein conservation scores and ratios</title><p>The general idea of the ‘Conservation Score’ is that it represents the lengths of the regions of BLAST homology between the <italic>S. cerevisiae</italic> query and the <italic>Saccharomycotina</italic> subjects in the beginning, middle, and end 40-aminoacid windows. Each <italic>S. cerevisiae</italic> query sequence was separated into two equal halves, and then BLASTs were done on both halves against all of <italic>Saccharomycotina</italic> (omitting all submissions from the <italic>Saccharomyces</italic> genus). For all subject proteins with high homology over any region (i.e. a bit-score greater than 50), one finds the pair-wise regions of homology with a BLAST ‘Alignment Score’ of 200 or more (red colored regions in the BLAST website ‘Graphic Summary’). The length of the high-homology region in the window of interest (but not the actual number of amino acid sequence matches within that region) contributes to the Conservation Score. That is, amino acids have to be within a region of homology found by BLAST in order to contribute to the score; even though all proteins begin with ‘M,’ these only contribute to the conservation score if they are within a region of BLAST homology. The Conservation Score is the sum of (the length of the homology in the window, multiplied by the proportion of qualified subject proteins with that length of homology). Example Conservation Scores are shown in <xref ref-type="supplementary-material" rid="supp3">Supplementary file 3</xref>, and an example Conservation Score is calculated in <xref ref-type="supplementary-material" rid="supp4">Supplementary file 4</xref>. We only considered conservation scores from proteins that had homology with at least 40 unique species in <italic>Saccharomycotina</italic>. We also omitted proteins that were shorter than 100 amino acids. As shown in <xref ref-type="supplementary-material" rid="supp3">Supplementary file 3</xref>, conservation scores ranged from 0, meaning no BLAST homology region within the window of 40 amino acids for any qualifying homolog in <italic>Saccharomycotina</italic>, up to a maximum of 40, meaning that all qualifying homologs in <italic>Saccharomycotina</italic> had matches starting at the first amino acid. In total, protein conservation analyses used 3964 <italic>S</italic>. <italic>cerevisiae</italic> proteins with high homology hits for BLAST done on the first and second half of proteins (<xref ref-type="fig" rid="fig4">Figure 4</xref>).</p></sec><sec id="s4-6"><title>Design of the fluorescent reporter gene constructs</title><p>We created a reporter gene based on reporter plasmid EKD1024 (<xref ref-type="bibr" rid="bib5">Brule et al., 2016</xref>). Briefly, a bidirectional galactose promoter simultaneously induces the expression of GFP and RFP in the presence of galactose, and we integrated the reporter into the yeast genome at the <italic>ADE2</italic> locus. We recoded GFP to give the first 41 codons of GFP a slow initial translation speed (SIT), medium initial translation speed (MIT), or fast initial translation speed (FIT), while maintaining the same amino acid sequence (<xref ref-type="fig" rid="fig8s1">Figure 8—figure supplement 1</xref>, sequences in <xref ref-type="supplementary-material" rid="supp5">Supplementary file 5</xref>). We also designed three more constructs (<xref ref-type="fig" rid="fig8s1">Figure 8—figure supplement 1</xref>, sequences in <xref ref-type="supplementary-material" rid="supp5">Supplementary file 5</xref>) with a SIT, MIT, or FIT upstream of one of the slowest and rarest codon pairs, CGA-CGG (replacing AAT-GAT, Asn-Asp), which is known to greatly attenuate gene expression in living yeast (<xref ref-type="bibr" rid="bib13">Gamble et al., 2016</xref>). Other rare codon pairs (CGA-CGA and CGA-CCG) have been shown to promote ribosome stalling (<xref ref-type="bibr" rid="bib31">Tesina et al., 2020</xref>) so in <xref ref-type="fig" rid="fig8">Figure 8</xref>, right, CGA-CGG operates as a putative ribosome collision site (PCS). Instead of a PCS, the constructs in <xref ref-type="fig" rid="fig8">Figure 8</xref>, left, had AAT-GAT (Asn Asp), a frequent codon pair with above average translation speed. The Relative Initial Translation Speed scores of the constructs were: SIT was 0.208; MIT was –0.0004; FIT was –0.166; SIT + PCS was 0.198; MIT + PCS was –0.0109; and FIT + PCS was –0.177. (Note that the RIT scores of the constructs with the PCS change because the PCS makes the translation speed of the body of the gene slower; that is, the change is due to a change in the denominator.)</p></sec><sec id="s4-7"><title>Yeast strains</title><p>The constructs were transformed into BY4741 (<italic>MATa his3Δ1 leu2Δ0 met15Δ0 ura3Δ0</italic>). The reporter expresses <italic>MET15</italic> allowing selection. Transformants were selected for Met+ on HULA plates (0.075 g/L Histidine; 0.075 g/L Uracil; 0.25 g/L Leucine; 0.075 g/L Adenine; 20 g/L D-Glucose; 5 g/L Ammonium Sulfate; 1.7 g/L Yeast Nitrogen Base). The reporter gene integrates into the <italic>ADE2</italic> locus, and thus successful transformants are <italic>ade2</italic>-delete which becomes red when grown on YPD plates; this was used as a secondary biological marker to confirm successful transformants. About 30 Met+, Ade-, red transformants were chosen for each recoded GFP construct. These transformants were pre-screened using a flow cytometer for absolute levels of galactose-induced green and red fluorescence; out of all the individual colonies initially chosen (~180), about 5 were rejected because their absolute levels of both GFP and RFP fluorescence were about twice as high as for other strains. We believe these rejected transformants contained two copies of the reporter construct. For each construct, 25 Met+, Ade-, red transformants were chosen for analysis. Strains are available upon request to BF, and sequences of SIT, MIT, and FIT constructs are available in <xref ref-type="supplementary-material" rid="supp5">Supplementary file 5</xref>.</p></sec><sec id="s4-8"><title>Flow cytometry analysis</title><p>The strains were inoculated in liquid HULA media, with 2% galactose, until mid-log phase (around 10–14 hr) containing around 3×10^7 cells/mL. The strains were sonicated to separate cells and the strains were stored on ice (typically around an hour) until data collection. An LSR Fortessa Flow Cytometer was used to measure GFP levels (All Events FITC-A Mean) and RFP levels (All Events PE-Texas Red-A Mean) across 75,000 events. As a control, a strain lacking a fluorescent reporter was used. For all samples, GFP levels were normalized by RFP levels. All samples (25 per experiment) were included in the analysis with no exclusions.</p></sec><sec id="s4-9"><title>Statistical tests</title><p>None of our analyses made assumptions regarding the normality of the data. As such, we only performed nonparametric statistics. Wilcoxon signed-rank tests were done for relevant pairwise analyses; when necessary, p-values were corrected for multiple comparisons using the Holm–Bonferroni method. Spearman correlations were used. We used the Kolmogorov–Smirnov goodness of fit test to confirm that the three distributions were significantly different (p&lt;0.001) for <xref ref-type="fig" rid="fig4">Figure 4</xref>.</p></sec></sec></body><back><sec sec-type="additional-information" id="s5"><title>Additional information</title><fn-group content-type="competing-interest"><title>Competing interests</title><fn fn-type="COI-statement" id="conf1"><p>No competing interests declared</p></fn></fn-group><fn-group content-type="author-contribution"><title>Author contributions</title><fn fn-type="con" id="con1"><p>Conceptualization, Data curation, Software, Formal analysis, Validation, Investigation, Visualization, Methodology, Writing - original draft, Project administration, Writing - review and editing</p></fn><fn fn-type="con" id="con2"><p>Conceptualization, Formal analysis, Supervision, Validation, Investigation, Visualization, Methodology, Writing - review and editing</p></fn><fn fn-type="con" id="con3"><p>Software, Formal analysis, Investigation, Methodology, Writing - review and editing</p></fn><fn fn-type="con" id="con4"><p>Conceptualization, Formal analysis, Supervision, Funding acquisition, Validation, Investigation, Visualization, Methodology, Writing - original draft, Project administration, Writing - review and editing</p></fn></fn-group></sec><sec sec-type="supplementary-material" id="s6"><title>Additional files</title><supplementary-material id="supp1"><label>Supplementary file 1.</label><caption><title>Ribosome residence time (RRT) values.</title><p>See attached Excel spreadsheet.</p></caption><media xlink:href="elife-89656-supp1-v1.xlsx" mimetype="application" mime-subtype="xlsx"/></supplementary-material><supplementary-material id="supp2"><label>Supplementary file 2.</label><caption><title>RRT statistics of various 5’ ends.</title><p>See attached Excel Spreadsheet.</p></caption><media xlink:href="elife-89656-supp2-v1.xlsx" mimetype="application" mime-subtype="xlsx"/></supplementary-material><supplementary-material id="supp3"><label>Supplementary file 3.</label><caption><title>Example conservation scores.</title><p>Scores were calculated as described in Materials and methods, but in this example, only for a subset of <italic>Saccharomycotina</italic>. ‘Total Hits’ is the number of different proteins from the sub-phylum <italic>Saccharomycotina</italic> subset giving a BLAST bit-score of at least 50. ‘Hits in the first 40 amino acids’ is the number of proteins (out of the proteins in the ‘Total Hits’ columns) that had a BLAST alignment with an alignment score &gt;200 matching any part of the first 40 amino acids of the query sequence (i.e. of <italic>PCA1</italic>, <italic>NSR1</italic>, etc.). ‘Query Start’ is the range of amino acid positions in the Query protein where the BLAST alignments started. For instance, for <italic>BUD5</italic>, the 125 S<italic>accharomycotina</italic> homologs had BLAST alignments that started at positions between amino acid 211 and amino acid 420 on <italic>S. cerevisiae BUD5</italic>; none had an alignment starting within the first 40 amino acids. For <italic>SNX41</italic>, 65 of the 121 hits had an alignment beginning within the first 40 amino acids of <italic>S. cerevisiae SNX41</italic>. For <italic>RPL12B</italic>, all 121 of the <italic>Saccharomycotina</italic> homologs had BLAST alignments starting at amino acid 1 of <italic>S. cerevisiae RPL12B</italic>. The ‘Conservation Score’ is the score calculated as described in Materials and methods. Note that the number of hits varies in part because the genomes of the <italic>Saccharomycotina</italic> species were not all fully sequenced. Thus, <italic>BNA2</italic> likely has fifteen fewer hits than <italic>TRP3</italic> because the <italic>BNA2</italic> locus was not sequenced in some species. However, the number of hits does not affect the conservation score, as long as the number meets the qualifying minimum.</p></caption><media xlink:href="elife-89656-supp3-v1.docx" mimetype="application" mime-subtype="docx"/></supplementary-material><supplementary-material id="supp4"><label>Supplementary file 4.</label><caption><title>Calculation of a conservation score.</title><p>‘Query start position in BLAST alignment’ is the amino acid residue of the <italic>S. cerevisiae</italic> query protein where a BLAST alignment (alignment score &gt;200) begins with a protein of <italic>Saccharomycotina</italic>. ‘Proportion of hits with this Q-Start Position’ is the proportion of qualifying <italic>Saccharomycotina</italic> hits (i.e. bit-score &gt;50) that have their alignment begin at this position. ‘Weight’ is multiplied by ‘Proportion,’ and the sum is the conservation score.</p></caption><media xlink:href="elife-89656-supp4-v1.docx" mimetype="application" mime-subtype="docx"/></supplementary-material><supplementary-material id="supp5"><label>Supplementary file 5.</label><caption><title>Sequences of Ramp genes in <xref ref-type="fig" rid="fig8">Figure 8</xref>.</title></caption><media xlink:href="elife-89656-supp5-v1.docx" mimetype="application" mime-subtype="docx"/></supplementary-material><supplementary-material id="mdar"><label>MDAR checklist</label><media xlink:href="elife-89656-mdarchecklist1-v1.docx" mimetype="application" mime-subtype="docx"/></supplementary-material><supplementary-material id="scode1"><label>Source code 1.</label><caption><title>We provide the custom R code written for this project as a text file, Source Code File 1.</title></caption><media xlink:href="elife-89656-code1-v1.zip" mimetype="application" mime-subtype="zip"/></supplementary-material></sec><sec sec-type="data-availability" id="s7"><title>Data availability</title><p>All data generated or analysed during this study are included in the manuscript and supporting files.</p></sec><ack id="ack"><title>Acknowledgements</title><p>We thank Steve Ketchum and Sangeet Honey for discussions that helped form the central idea of this manuscript.</p></ack><ref-list><title>References</title><ref id="bib1"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Arava</surname><given-names>Y</given-names></name><name><surname>Wang</surname><given-names>Y</given-names></name><name><surname>Storey</surname><given-names>JD</given-names></name><name><surname>Liu</surname><given-names>CL</given-names></name><name><surname>Brown</surname><given-names>PO</given-names></name><name><surname>Herschlag</surname><given-names>D</given-names></name></person-group><year iso-8601-date="2003">2003</year><article-title>Genome-wide analysis of mRNA translation profiles in <italic>Saccharomyces cerevisiae</italic></article-title><source>PNAS</source><volume>100</volume><fpage>3889</fpage><lpage>3894</lpage><pub-id pub-id-type="doi">10.1073/pnas.0635171100</pub-id><pub-id pub-id-type="pmid">12660367</pub-id></element-citation></ref><ref id="bib2"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Bazykin</surname><given-names>GA</given-names></name><name><surname>Kochetov</surname><given-names>AV</given-names></name></person-group><year iso-8601-date="2011">2011</year><article-title>Alternative translation start sites are conserved in eukaryotic genomes</article-title><source>Nucleic Acids Research</source><volume>39</volume><fpage>567</fpage><lpage>577</lpage><pub-id pub-id-type="doi">10.1093/nar/gkq806</pub-id><pub-id pub-id-type="pmid">20864444</pub-id></element-citation></ref><ref id="bib3"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Bonderoff</surname><given-names>JM</given-names></name><name><surname>Lloyd</surname><given-names>RE</given-names></name></person-group><year iso-8601-date="2010">2010</year><article-title>Time-dependent increase in ribosome processivity</article-title><source>Nucleic Acids Research</source><volume>38</volume><fpage>7054</fpage><lpage>7067</lpage><pub-id pub-id-type="doi">10.1093/nar/gkq566</pub-id><pub-id pub-id-type="pmid">20571082</pub-id></element-citation></ref><ref id="bib4"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Bricout</surname><given-names>R</given-names></name><name><surname>Weil</surname><given-names>D</given-names></name><name><surname>Stroebel</surname><given-names>D</given-names></name><name><surname>Genovesio</surname><given-names>A</given-names></name><name><surname>Roest Crollius</surname><given-names>H</given-names></name></person-group><year iso-8601-date="2023">2023</year><article-title>Evolution is not uniform along coding sequences</article-title><source>Molecular Biology and Evolution</source><volume>40</volume><elocation-id>msad042</elocation-id><pub-id pub-id-type="doi">10.1093/molbev/msad042</pub-id><pub-id pub-id-type="pmid">36857092</pub-id></element-citation></ref><ref id="bib5"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Brule</surname><given-names>CE</given-names></name><name><surname>Dean</surname><given-names>KM</given-names></name><name><surname>Grayhack</surname><given-names>EJ</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>RNA-ID, a powerful tool for identifying and characterizing regulatory sequences</article-title><source>Methods in Enzymology</source><volume>572</volume><fpage>237</fpage><lpage>253</lpage><pub-id pub-id-type="doi">10.1016/bs.mie.2016.02.003</pub-id><pub-id pub-id-type="pmid">27241757</pub-id></element-citation></ref><ref id="bib6"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Burkhardt</surname><given-names>DH</given-names></name><name><surname>Rouskin</surname><given-names>S</given-names></name><name><surname>Zhang</surname><given-names>Y</given-names></name><name><surname>Li</surname><given-names>GW</given-names></name><name><surname>Weissman</surname><given-names>JS</given-names></name><name><surname>Gross</surname><given-names>CA</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>Operon mRNAs are organized into ORF-centric structures that predict translation efficiency</article-title><source>eLife</source><volume>6</volume><elocation-id>e22037</elocation-id><pub-id pub-id-type="doi">10.7554/eLife.22037</pub-id><pub-id pub-id-type="pmid">28139975</pub-id></element-citation></ref><ref id="bib7"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Cope</surname><given-names>AL</given-names></name><name><surname>Hettich</surname><given-names>RL</given-names></name><name><surname>Gilchrist</surname><given-names>MA</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Quantifying codon usage in signal peptides: Gene expression and amino acid usage explain apparent selection for inefficient codons</article-title><source>Biochimica et Biophysica Acta. Biomembranes</source><volume>1860</volume><fpage>2479</fpage><lpage>2485</lpage><pub-id pub-id-type="doi">10.1016/j.bbamem.2018.09.010</pub-id><pub-id pub-id-type="pmid">30279149</pub-id></element-citation></ref><ref id="bib8"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Cuperus</surname><given-names>JT</given-names></name><name><surname>Groves</surname><given-names>B</given-names></name><name><surname>Kuchina</surname><given-names>A</given-names></name><name><surname>Rosenberg</surname><given-names>AB</given-names></name><name><surname>Jojic</surname><given-names>N</given-names></name><name><surname>Fields</surname><given-names>S</given-names></name><name><surname>Seelig</surname><given-names>G</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>Deep learning of the regulatory grammar of yeast 5’ untranslated regions from 500,000 random sequences</article-title><source>Genome Research</source><volume>27</volume><fpage>2015</fpage><lpage>2024</lpage><pub-id pub-id-type="doi">10.1101/gr.224964.117</pub-id><pub-id pub-id-type="pmid">29097404</pub-id></element-citation></ref><ref id="bib9"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Dao Duc</surname><given-names>K</given-names></name><name><surname>Song</surname><given-names>YS</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>The impact of ribosomal interference, codon usage, and exit tunnel interactions on translation elongation rate variation</article-title><source>PLOS Genetics</source><volume>14</volume><elocation-id>e1007166</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pgen.1007166</pub-id><pub-id pub-id-type="pmid">29337993</pub-id></element-citation></ref><ref id="bib10"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Dean</surname><given-names>KM</given-names></name><name><surname>Grayhack</surname><given-names>EJ</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>RNA-ID, a highly sensitive and robust method to identify cis-regulatory sequences using superfolder GFP and a fluorescence-based assay</article-title><source>RNA</source><volume>18</volume><fpage>2335</fpage><lpage>2344</lpage><pub-id pub-id-type="doi">10.1261/rna.035907.112</pub-id><pub-id pub-id-type="pmid">23097427</pub-id></element-citation></ref><ref id="bib11"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Eisenberg</surname><given-names>AR</given-names></name><name><surname>Higdon</surname><given-names>AL</given-names></name><name><surname>Hollerer</surname><given-names>I</given-names></name><name><surname>Fields</surname><given-names>AP</given-names></name><name><surname>Jungreis</surname><given-names>I</given-names></name><name><surname>Diamond</surname><given-names>PD</given-names></name><name><surname>Kellis</surname><given-names>M</given-names></name><name><surname>Jovanovic</surname><given-names>M</given-names></name><name><surname>Brar</surname><given-names>GA</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>Translation initiation site profiling reveals widespread synthesis of non-aug-initiated protein isoforms in yeast</article-title><source>Cell Systems</source><volume>11</volume><fpage>145</fpage><lpage>160</lpage><pub-id pub-id-type="doi">10.1016/j.cels.2020.06.011</pub-id><pub-id pub-id-type="pmid">32710835</pub-id></element-citation></ref><ref id="bib12"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Fukasawa</surname><given-names>Y</given-names></name><name><surname>Tsuji</surname><given-names>J</given-names></name><name><surname>Fu</surname><given-names>SC</given-names></name><name><surname>Tomii</surname><given-names>K</given-names></name><name><surname>Horton</surname><given-names>P</given-names></name><name><surname>Imai</surname><given-names>K</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>MitoFates: improved prediction of mitochondrial targeting sequences and their cleavage sites</article-title><source>Molecular &amp; Cellular Proteomics</source><volume>14</volume><fpage>1113</fpage><lpage>1126</lpage><pub-id pub-id-type="doi">10.1074/mcp.M114.043083</pub-id><pub-id pub-id-type="pmid">25670805</pub-id></element-citation></ref><ref id="bib13"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Gamble</surname><given-names>CE</given-names></name><name><surname>Brule</surname><given-names>CE</given-names></name><name><surname>Dean</surname><given-names>KM</given-names></name><name><surname>Fields</surname><given-names>S</given-names></name><name><surname>Grayhack</surname><given-names>EJ</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Adjacent codons act in concert to modulate translation efficiency in yeast</article-title><source>Cell</source><volume>166</volume><fpage>679</fpage><lpage>690</lpage><pub-id pub-id-type="doi">10.1016/j.cell.2016.05.070</pub-id><pub-id pub-id-type="pmid">27374328</pub-id></element-citation></ref><ref id="bib14"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Gardin</surname><given-names>J</given-names></name><name><surname>Yeasmin</surname><given-names>R</given-names></name><name><surname>Yurovsky</surname><given-names>A</given-names></name><name><surname>Cai</surname><given-names>Y</given-names></name><name><surname>Skiena</surname><given-names>S</given-names></name><name><surname>Futcher</surname><given-names>B</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>Measurement of average decoding rates of the 61 sense codons in vivo</article-title><source>eLife</source><volume>3</volume><elocation-id>e03735</elocation-id><pub-id pub-id-type="doi">10.7554/eLife.03735</pub-id><pub-id pub-id-type="pmid">25347064</pub-id></element-citation></ref><ref id="bib15"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Gould</surname><given-names>SJ</given-names></name><name><surname>Lewontin</surname><given-names>RC</given-names></name></person-group><year iso-8601-date="1979">1979</year><article-title>The spandrels of San Marco and the Panglossian paradigm: a critique of the adaptationist programme</article-title><source>Proceedings of the Royal Society of London. Series B, Biological Sciences</source><volume>205</volume><fpage>581</fpage><lpage>598</lpage><pub-id pub-id-type="doi">10.1098/rspb.1979.0086</pub-id><pub-id pub-id-type="pmid">42062</pub-id></element-citation></ref><ref id="bib16"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Gritsenko</surname><given-names>AA</given-names></name><name><surname>Hulsman</surname><given-names>M</given-names></name><name><surname>Reinders</surname><given-names>MJT</given-names></name><name><surname>de Ridder</surname><given-names>D</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>Unbiased quantitative models of protein translation derived from ribosome profiling data</article-title><source>PLOS Computational Biology</source><volume>11</volume><elocation-id>e1004336</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pcbi.1004336</pub-id><pub-id pub-id-type="pmid">26275099</pub-id></element-citation></ref><ref id="bib17"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Gu</surname><given-names>W</given-names></name><name><surname>Zhou</surname><given-names>T</given-names></name><name><surname>Wilke</surname><given-names>CO</given-names></name></person-group><year iso-8601-date="2010">2010</year><article-title>A universal trend of reduced mRNA stability near the translation-initiation site in prokaryotes and eukaryotes</article-title><source>PLOS Computational Biology</source><volume>6</volume><elocation-id>e1000664</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pcbi.1000664</pub-id><pub-id pub-id-type="pmid">20140241</pub-id></element-citation></ref><ref id="bib18"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hall</surname><given-names>MN</given-names></name><name><surname>Gabay</surname><given-names>J</given-names></name><name><surname>Débarbouillé</surname><given-names>M</given-names></name><name><surname>Schwartz</surname><given-names>M</given-names></name></person-group><year iso-8601-date="1982">1982</year><article-title>A role for mRNA secondary structure in the control of translation initiation</article-title><source>Nature</source><volume>295</volume><fpage>616</fpage><lpage>618</lpage><pub-id pub-id-type="doi">10.1038/295616a0</pub-id><pub-id pub-id-type="pmid">6799842</pub-id></element-citation></ref><ref id="bib19"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ingolia</surname><given-names>NT</given-names></name><name><surname>Ghaemmaghami</surname><given-names>S</given-names></name><name><surname>Newman</surname><given-names>JRS</given-names></name><name><surname>Weissman</surname><given-names>JS</given-names></name></person-group><year iso-8601-date="2009">2009</year><article-title>Genome-wide analysis in vivo of translation with nucleotide resolution using ribosome profiling</article-title><source>Science</source><volume>324</volume><fpage>218</fpage><lpage>223</lpage><pub-id pub-id-type="doi">10.1126/science.1168978</pub-id><pub-id pub-id-type="pmid">19213877</pub-id></element-citation></ref><ref id="bib20"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Jan</surname><given-names>CH</given-names></name><name><surname>Williams</surname><given-names>CC</given-names></name><name><surname>Weissman</surname><given-names>JS</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>Principles of ER cotranslational translocation revealed by proximity-specific ribosome profiling</article-title><source>Science</source><volume>346</volume><elocation-id>1257521</elocation-id><pub-id pub-id-type="doi">10.1126/science.1257521</pub-id><pub-id pub-id-type="pmid">25378630</pub-id></element-citation></ref><ref id="bib21"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kochetov</surname><given-names>AV</given-names></name></person-group><year iso-8601-date="2008">2008</year><article-title>Alternative translation start sites and hidden coding potential of eukaryotic mRNAs</article-title><source>BioEssays</source><volume>30</volume><fpage>683</fpage><lpage>691</lpage><pub-id pub-id-type="doi">10.1002/bies.20771</pub-id><pub-id pub-id-type="pmid">18536038</pub-id></element-citation></ref><ref id="bib22"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kudla</surname><given-names>G</given-names></name><name><surname>Murray</surname><given-names>AW</given-names></name><name><surname>Tollervey</surname><given-names>D</given-names></name><name><surname>Plotkin</surname><given-names>JB</given-names></name></person-group><year iso-8601-date="2009">2009</year><article-title>Coding-sequence determinants of gene expression in <italic>Escherichia coli</italic></article-title><source>Science</source><volume>324</volume><fpage>255</fpage><lpage>258</lpage><pub-id pub-id-type="doi">10.1126/science.1170160</pub-id><pub-id pub-id-type="pmid">19359587</pub-id></element-citation></ref><ref id="bib23"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Lareau</surname><given-names>LF</given-names></name><name><surname>Hite</surname><given-names>DH</given-names></name><name><surname>Hogan</surname><given-names>GJ</given-names></name><name><surname>Brown</surname><given-names>PO</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>Distinct stages of the translation elongation cycle revealed by sequencing ribosome-protected mRNA fragments</article-title><source>eLife</source><volume>3</volume><elocation-id>e01257</elocation-id><pub-id pub-id-type="doi">10.7554/eLife.01257</pub-id><pub-id pub-id-type="pmid">24842990</pub-id></element-citation></ref><ref id="bib24"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Lipson</surname><given-names>D</given-names></name><name><surname>Raz</surname><given-names>T</given-names></name><name><surname>Kieu</surname><given-names>A</given-names></name><name><surname>Jones</surname><given-names>DR</given-names></name><name><surname>Giladi</surname><given-names>E</given-names></name><name><surname>Thayer</surname><given-names>E</given-names></name><name><surname>Thompson</surname><given-names>JF</given-names></name><name><surname>Letovsky</surname><given-names>S</given-names></name><name><surname>Milos</surname><given-names>P</given-names></name><name><surname>Causey</surname><given-names>M</given-names></name></person-group><year iso-8601-date="2009">2009</year><article-title>Quantification of the yeast transcriptome by single-molecule sequencing</article-title><source>Nature Biotechnology</source><volume>27</volume><fpage>652</fpage><lpage>658</lpage><pub-id pub-id-type="doi">10.1038/nbt.1551</pub-id><pub-id pub-id-type="pmid">19581875</pub-id></element-citation></ref><ref id="bib25"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Nackley</surname><given-names>AG</given-names></name><name><surname>Shabalina</surname><given-names>SA</given-names></name><name><surname>Tchivileva</surname><given-names>IE</given-names></name><name><surname>Satterfield</surname><given-names>K</given-names></name><name><surname>Korchynskyi</surname><given-names>O</given-names></name><name><surname>Makarov</surname><given-names>SS</given-names></name><name><surname>Maixner</surname><given-names>W</given-names></name><name><surname>Diatchenko</surname><given-names>L</given-names></name></person-group><year iso-8601-date="2006">2006</year><article-title>Human catechol-O-methyltransferase haplotypes modulate protein expression by altering mRNA secondary structure</article-title><source>Science</source><volume>314</volume><fpage>1930</fpage><lpage>1933</lpage><pub-id pub-id-type="doi">10.1126/science.1131262</pub-id><pub-id pub-id-type="pmid">17185601</pub-id></element-citation></ref><ref id="bib26"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Pechmann</surname><given-names>S</given-names></name><name><surname>Chartron</surname><given-names>JW</given-names></name><name><surname>Frydman</surname><given-names>J</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>Local slowdown of translation by nonoptimal codons promotes nascent-chain recognition by SRP in vivo</article-title><source>Nature Structural &amp; Molecular Biology</source><volume>21</volume><fpage>1100</fpage><lpage>1105</lpage><pub-id pub-id-type="doi">10.1038/nsmb.2919</pub-id><pub-id pub-id-type="pmid">25420103</pub-id></element-citation></ref><ref id="bib27"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Pédelacq</surname><given-names>J-D</given-names></name><name><surname>Cabantous</surname><given-names>S</given-names></name><name><surname>Tran</surname><given-names>T</given-names></name><name><surname>Terwilliger</surname><given-names>TC</given-names></name><name><surname>Waldo</surname><given-names>GS</given-names></name></person-group><year iso-8601-date="2006">2006</year><article-title>Engineering and characterization of a superfolder green fluorescent protein</article-title><source>Nature Biotechnology</source><volume>24</volume><fpage>79</fpage><lpage>88</lpage><pub-id pub-id-type="doi">10.1038/nbt1172</pub-id><pub-id pub-id-type="pmid">16369541</pub-id></element-citation></ref><ref id="bib28"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Presnyak</surname><given-names>V</given-names></name><name><surname>Alhusaini</surname><given-names>N</given-names></name><name><surname>Chen</surname><given-names>YH</given-names></name><name><surname>Martin</surname><given-names>S</given-names></name><name><surname>Morris</surname><given-names>N</given-names></name><name><surname>Kline</surname><given-names>N</given-names></name><name><surname>Olson</surname><given-names>S</given-names></name><name><surname>Weinberg</surname><given-names>D</given-names></name><name><surname>Baker</surname><given-names>KE</given-names></name><name><surname>Graveley</surname><given-names>BR</given-names></name><name><surname>Coller</surname><given-names>J</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>Codon optimality is a major determinant of mRNA stability</article-title><source>Cell</source><volume>160</volume><fpage>1111</fpage><lpage>1124</lpage><pub-id pub-id-type="doi">10.1016/j.cell.2015.02.029</pub-id><pub-id pub-id-type="pmid">25768907</pub-id></element-citation></ref><ref id="bib29"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Shah</surname><given-names>P</given-names></name><name><surname>Ding</surname><given-names>Y</given-names></name><name><surname>Niemczyk</surname><given-names>M</given-names></name><name><surname>Kudla</surname><given-names>G</given-names></name><name><surname>Plotkin</surname><given-names>JB</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>Rate-limiting steps in yeast protein translation</article-title><source>Cell</source><volume>153</volume><fpage>1589</fpage><lpage>1601</lpage><pub-id pub-id-type="doi">10.1016/j.cell.2013.05.049</pub-id><pub-id pub-id-type="pmid">23791185</pub-id></element-citation></ref><ref id="bib30"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Sharma</surname><given-names>AK</given-names></name><name><surname>Sormanni</surname><given-names>P</given-names></name><name><surname>Ahmed</surname><given-names>N</given-names></name><name><surname>Ciryam</surname><given-names>P</given-names></name><name><surname>Friedrich</surname><given-names>UA</given-names></name><name><surname>Kramer</surname><given-names>G</given-names></name><name><surname>O’Brien</surname><given-names>EP</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>A chemical kinetic basis for measuring translation initiation and elongation rates from ribosome profiling data</article-title><source>PLOS Computational Biology</source><volume>15</volume><elocation-id>e1007070</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pcbi.1007070</pub-id><pub-id pub-id-type="pmid">31120880</pub-id></element-citation></ref><ref id="bib31"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Tesina</surname><given-names>P</given-names></name><name><surname>Lessen</surname><given-names>LN</given-names></name><name><surname>Buschauer</surname><given-names>R</given-names></name><name><surname>Cheng</surname><given-names>J</given-names></name><name><surname>Wu</surname><given-names>CC-C</given-names></name><name><surname>Berninghausen</surname><given-names>O</given-names></name><name><surname>Buskirk</surname><given-names>AR</given-names></name><name><surname>Becker</surname><given-names>T</given-names></name><name><surname>Beckmann</surname><given-names>R</given-names></name><name><surname>Green</surname><given-names>R</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>Molecular mechanism of translational stalling by inhibitory codon combinations and poly(A) tracts</article-title><source>The EMBO Journal</source><volume>39</volume><elocation-id>e103365</elocation-id><pub-id pub-id-type="doi">10.15252/embj.2019103365</pub-id><pub-id pub-id-type="pmid">31858614</pub-id></element-citation></ref><ref id="bib32"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Tuller</surname><given-names>T</given-names></name><name><surname>Carmi</surname><given-names>A</given-names></name><name><surname>Vestsigian</surname><given-names>K</given-names></name><name><surname>Navon</surname><given-names>S</given-names></name><name><surname>Dorfan</surname><given-names>Y</given-names></name><name><surname>Zaborske</surname><given-names>J</given-names></name><name><surname>Pan</surname><given-names>T</given-names></name><name><surname>Dahan</surname><given-names>O</given-names></name><name><surname>Furman</surname><given-names>I</given-names></name><name><surname>Pilpel</surname><given-names>Y</given-names></name></person-group><year iso-8601-date="2010">2010</year><article-title>An evolutionarily conserved mechanism for controlling the efficiency of protein translation</article-title><source>Cell</source><volume>141</volume><fpage>344</fpage><lpage>354</lpage><pub-id pub-id-type="doi">10.1016/j.cell.2010.03.031</pub-id><pub-id pub-id-type="pmid">20403328</pub-id></element-citation></ref><ref id="bib33"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Tunney</surname><given-names>R</given-names></name><name><surname>McGlincy</surname><given-names>NJ</given-names></name><name><surname>Graham</surname><given-names>ME</given-names></name><name><surname>Naddaf</surname><given-names>N</given-names></name><name><surname>Pachter</surname><given-names>L</given-names></name><name><surname>Lareau</surname><given-names>LF</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Accurate design of translational output by a neural network model of ribosome distribution</article-title><source>Nature Structural &amp; Molecular Biology</source><volume>25</volume><fpage>577</fpage><lpage>582</lpage><pub-id pub-id-type="doi">10.1038/s41594-018-0080-2</pub-id><pub-id pub-id-type="pmid">29967537</pub-id></element-citation></ref><ref id="bib34"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Verma</surname><given-names>M</given-names></name><name><surname>Choi</surname><given-names>J</given-names></name><name><surname>Cottrell</surname><given-names>KA</given-names></name><name><surname>Lavagnino</surname><given-names>Z</given-names></name><name><surname>Thomas</surname><given-names>EN</given-names></name><name><surname>Pavlovic-Djuranovic</surname><given-names>S</given-names></name><name><surname>Szczesny</surname><given-names>P</given-names></name><name><surname>Piston</surname><given-names>DW</given-names></name><name><surname>Zaher</surname><given-names>HS</given-names></name><name><surname>Puglisi</surname><given-names>JD</given-names></name><name><surname>Djuranovic</surname><given-names>S</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>A short translational ramp determines the efficiency of protein synthesis</article-title><source>Nature Communications</source><volume>10</volume><elocation-id>5774</elocation-id><pub-id pub-id-type="doi">10.1038/s41467-019-13810-1</pub-id><pub-id pub-id-type="pmid">31852903</pub-id></element-citation></ref><ref id="bib35"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname><given-names>H</given-names></name><name><surname>McManus</surname><given-names>J</given-names></name><name><surname>Kingsford</surname><given-names>C</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>Accurate recovery of ribosome positions reveals slow translation of wobble-pairing codons in yeast</article-title><source>Journal of Computational Biology</source><volume>24</volume><fpage>486</fpage><lpage>500</lpage><pub-id pub-id-type="doi">10.1089/cmb.2016.0147</pub-id><pub-id pub-id-type="pmid">27726445</pub-id></element-citation></ref><ref id="bib36"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Weinberg</surname><given-names>DE</given-names></name><name><surname>Shah</surname><given-names>P</given-names></name><name><surname>Eichhorn</surname><given-names>SW</given-names></name><name><surname>Hussmann</surname><given-names>JA</given-names></name><name><surname>Plotkin</surname><given-names>JB</given-names></name><name><surname>Bartel</surname><given-names>DP</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Improved ribosome-footprint and mRNA measurements provide insights into dynamics and regulation of yeast translation</article-title><source>Cell Reports</source><volume>14</volume><fpage>1787</fpage><lpage>1799</lpage><pub-id pub-id-type="doi">10.1016/j.celrep.2016.01.043</pub-id><pub-id pub-id-type="pmid">26876183</pub-id></element-citation></ref><ref id="bib37"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Williams</surname><given-names>CC</given-names></name><name><surname>Jan</surname><given-names>CH</given-names></name><name><surname>Weissman</surname><given-names>JS</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>Targeting and plasticity of mitochondrial proteins revealed by proximity-specific ribosome profiling</article-title><source>Science</source><volume>346</volume><fpage>748</fpage><lpage>751</lpage><pub-id pub-id-type="doi">10.1126/science.1257522</pub-id><pub-id pub-id-type="pmid">25378625</pub-id></element-citation></ref><ref id="bib38"><element-citation publication-type="preprint"><person-group person-group-type="author"><name><surname>Yurovsky</surname><given-names>A</given-names></name><name><surname>Gardin</surname><given-names>J</given-names></name><name><surname>Futcher</surname><given-names>B</given-names></name><name><surname>Skiena</surname><given-names>S</given-names></name></person-group><year iso-8601-date="2022">2022</year><article-title>A statistical detector for ribosomal frameshifts and dual encodings based on ribosome profiling</article-title><source>bioRxiv</source><pub-id pub-id-type="doi">10.1101/2022.06.06.495024</pub-id></element-citation></ref></ref-list></back><sub-article article-type="editor-report" id="sa0"><front-stub><article-id pub-id-type="doi">10.7554/eLife.89656.3.sa0</article-id><title-group><article-title>eLife assessment</article-title></title-group><contrib-group><contrib contrib-type="author"><name><surname>Hinnebusch</surname><given-names>Alan G</given-names></name><role specific-use="editor">Reviewing Editor</role><aff><institution>Eunice Kennedy Shriver National Institute of Child Health and Human Development</institution><country>United States</country></aff></contrib></contrib-group><kwd-group kwd-group-type="evidence-strength"><kwd>Convincing</kwd><kwd>Incomplete</kwd></kwd-group><kwd-group kwd-group-type="claim-importance"><kwd>Important</kwd></kwd-group></front-stub><body><p>This is an <bold>important</bold> contribution to the origins and translational consequences of the relatively low rate of translation elongation in the first ∼30-50 codons of genes in most organisms. The authors provide <bold>convincing</bold> evidence that the prevalence of rare codons in the first ~40 codons in yeast is due to the relatively recent evolution of these coding sequences, or of lower purifying selection operating on them, and that a preponderance of codons encoded by rare tRNAs near the N-terminus is not associated with higher translational efficiency in the manner proposed by the &quot;translational ramp&quot; hypothesis. The work is <bold>incomplete</bold> in that the results of reporter assays may have been confounded by alterations of mRNA sequence or structure that could have influenced their translation or mRNA stability; that the work cannot fully account for a greater enrichment of slowly translated codons in N-terminal vs. C-terminal regions; and that the work does not resolve whether translation elongation through N-terminal coding is truly slow.</p></body></sub-article><sub-article article-type="referee-report" id="sa1"><front-stub><article-id pub-id-type="doi">10.7554/eLife.89656.3.sa1</article-id><title-group><article-title>Reviewer #1 (Public Review):</article-title></title-group><contrib-group><contrib contrib-type="author"><anonymous/><role specific-use="referee">Reviewer</role></contrib></contrib-group></front-stub><body><p>The manuscript by Sejour et al. is testing &quot;translational ramp&quot; model described previously by Tuller et al. in <italic>S. cerevisiae</italic>. Authors are using bioinformatics and reporter based experimental approaches to test whether &quot;rare codons&quot; in the first 40 codons of the gene coding sequences increase translation efficiency and regulate abundance of translation products in yeast cells. Authors conclude that &quot;translation ramp&quot; model does not have support using a new set of reporters and bioinformatics analyses. The strength of bioinformatic evidence and experimental analyses (even very limited) of the rare codons insertion in the reporter make a compelling case for the authors claims. However the major weakness of the manuscript is that authors do not take into account other models that previously disputed &quot;rare or slow codon&quot; model of Tuller et al. and overstate their own results that are rather limited. This maintains to be the weak part of the manuscript even in the revised form.</p><p>The studies that authors do not mention argue with &quot;translation ramp&quot; model and show more thorough analyses of translation initiation to elongation transition as well as early elongation &quot;slow down&quot; in ribosome profiling data. Moreover several studies have used bioinformatical analyses to point out the evolution of N-terminal sequences in multiple model organisms including yeast, focusing on either upstream ORFs (uORFs) or already annotated ORFs. The authors did not mention multiple of these studies in their revised manuscript and did not comment on their own results in the context of these previous studies. As such the authors approach to data presentation, writing and data discussion makes the manuscript rather biased, focused on criticizing Tuller et al. study and short on discussing multiple other possible reasons for slow translation elongation at the beginning of the protein synthesis. This all together makes the manuscript at the end very limited.</p></body></sub-article><sub-article article-type="referee-report" id="sa2"><front-stub><article-id pub-id-type="doi">10.7554/eLife.89656.3.sa2</article-id><title-group><article-title>Reviewer #2 (Public Review):</article-title></title-group><contrib-group><contrib contrib-type="author"><anonymous/><role specific-use="referee">Reviewer</role></contrib></contrib-group></front-stub><body><p>Tuller et al. first made the curious observation, that the first ∼30-50 codons in most organisms are encoded by scarce tRNAs and appear to be translated slower than the rest of the coding sequences (CDS). They speculated that this has evolved to pace ribosomes on CDS and prevent ribosome collisions during elongation - the &quot;Ramp&quot; hypothesis. Various aspects of this hypothesis, both factual and in terms of interpreting the results, have been challenged ever since. Sejour et al. present compelling results confirming the slower translation of the first ~40 codons in <italic>S. cerevisiae</italic> but providing an alternative explanation for this phenomenon. Specifically, they show that the higher amino acid sequence divergence of N-terminal ends of proteins and accompanying lower purifying selection (perhaps the result of de novo evolution) is sufficient to explain the prevalence of rare slow codons in these regions. These results are an important contribution in understanding how aspects of the evolution of protein coding regions can affect translation efficiency on these sequences and directly challenge the &quot;Ramp&quot; hypothesis proposed by Tuller et al.</p><p>I believe the data is presented clearly and the results generally justify the conclusions.</p></body></sub-article><sub-article article-type="author-comment" id="sa3"><front-stub><article-id pub-id-type="doi">10.7554/eLife.89656.3.sa3</article-id><title-group><article-title>Author response</article-title></title-group><contrib-group><contrib contrib-type="author"><name><surname>Sejour</surname><given-names>Richard</given-names></name><role specific-use="author">Author</role><aff><institution>Stony Brook University</institution><addr-line><named-content content-type="city">Stony Brook</named-content></addr-line><country>United States</country></aff></contrib><contrib contrib-type="author"><name><surname>Leatherwood</surname><given-names>Janet</given-names></name><role specific-use="author">Author</role><aff><institution>Stony Brook University</institution><addr-line><named-content content-type="city">Stony Brook</named-content></addr-line><country>United States</country></aff></contrib><contrib contrib-type="author"><name><surname>Yurovsky</surname><given-names>Alisa</given-names></name><role specific-use="author">Author</role><aff><institution>Stony Brook University</institution><addr-line><named-content content-type="city">Stony Brook</named-content></addr-line><country>United States</country></aff></contrib><contrib contrib-type="author"><name><surname>Futcher</surname><given-names>Bruce</given-names></name><role specific-use="author">Author</role><aff><institution>Stony Brook University</institution><addr-line><named-content content-type="city">Stony Brook</named-content></addr-line><country>United States</country></aff></contrib></contrib-group></front-stub><body><p>The following is the authors’ response to the original reviews.</p><p>Response to Reviewers:</p><p>We thank the reviewers for their comments, and their evident close reading of the manuscript.Generally, we agree with the reviewers on the strengths and weaknesses of our manuscript. Ourrevised manuscript has a more extensive discussion of alternative explanations for initial highribosome density as seen by ribosome profiling, and which more specifically points out thelimitations of our work.</p><p>As a preface to specific responses to the reviewers, we will say that we could divide observationsof slow initial translation into two categories, which we will call “encoded slow codons”, and“increased ribosome density”. With respect to the first category, Tuller et al. documented initial“encoded slow codons”, that is, there is a statistical excess of rare, slowly-translated codons atthe 5’ ends of genes. Although the size of this effect is small, statistical significance is extremelyhigh, and the existence of this enrichment is not in any doubt. At first sight, this appears to be astrong indication of a preference for slow initial translation. In our opinion, our maincontribution is to show that there is an alternative explanation for this initial enrichment of rare,slow codons—that they are a spandrel, a consequence of sequence plasticity at the 5’ (and 3’)ends of genes. The reviewers seem to generally agree with this, and we are not aware that anyother work has provided an explanation for the 5’ enrichment of rare codons.</p><p>The second category of observations pertaining to slow initial translation is “increased ribosomedensity”. Early ribosome profiling studies used cycloheximide to arrest cell growth, and thesestudies showed a higher density of ribosomes near the 5’ end of genes than elsewhere. This highinitial ribosome density helped motivate the paper of Tuller et al., though their finding of“encoded slow codons” could explain only a very small part of the increased ribosome density.More modern ribosome profiling studies do not use cycloheximide as the first step in arrestingtranslation, and in these studies, the density of ribosomes near the 5’ end of genes is greatlyreduced. And yet, there remains, even in the absence of cycloheximide at the first step, asignificantly increased density of ribosomes near the 5’ end (e.g., Weinberg et al., 2016).(However, most or all of these studies do use cycloheximide at a later step in the protocol, andthe possibility of a cycloheximide artefact is difficult to exclude.) Some of the reviewer’sconcerns are that we do not explain the increased 5’ ribosome density seen by ribosomeprofiling. We agree; but we feel it is not the main point of our manuscript. In revision, we moreextensively discuss other work on increased ribosome density, and more explicitly point out thelimitations of our manuscript in this regard. We also note, though, that increased ribosomedensity is not a direct measure of translation speed—it can have other causes.</p><p>Specific Responses.</p><p>Reviewer 1 was concerned that we did not more fully discuss other work on possible reasons forslow initial translation. We discuss such work more extensively in our revision. However, as faras we know, none of this work proposes a reason for the 5’ enrichment of rare, slow codons, andthis is the main point of our paper. Furthermore, it is not completely clear that there is any slowinitial translation. The increase in ribosome density seen in flash-freeze ribosome profiling couldbe an artefact of the use of cycloheximide at the thaw step of the protocols; or it could be a real measure of high ribosome density that occurs for some other reason than slow translation (e.g.,ribosomes might have low processivity at the 5’ end).</p><p>Reviewer 1 was also concerned about confounding effects in our reporter gene analysis of theeffects of different codons on efficiency of translation. We have two comments. First, it isimportant to remember that although we changed codons in our reporters, we did not change anyamino acids. We changed codons only to synonymous codons. Thus at least one of thereviewer’s possible confounding effects—interactions of the nascent peptide chain with the exitchannel of the ribosome—does not apply. However, of course, the mRNA nucleotide sequenceis altered, and this would cause a change in mRNA structure or abundance, which could matter.We agree this is a limitation to our approach. However, to fully address it, we feel it would benecessary to examine a really large number of quite different sequences, which is beyond thescope of this work. Furthermore, mRNAs with low secondary structure at the 5’ end probablyhave relatively high rates of initiation, and also relatively high rates of elongation, and it mightbe quite difficult to disentangle these. But in neither case is there an argument that slow initialtranslation is efficient. Accurate measurement of mRNA levels would be helpful, but would notdisentangle rates of initiation from rates of elongation as causes of changes in expression.</p><p>Reviewer 2 was concerned that the conservation scores for the 5’ 40 amino acids, and the 3’ 40amino acids were similar, but slow translation was only statistically significant for the 5’ 40amino acids. As we say in the manuscript, we are also puzzled by this. We note that 3’translation is statistically slow, if one looks over the last 100 amino acids. Our best effort at anexplanation is a sort of reverse-Tuller explanation: that in the last 40 amino acids, the new slowcodons created by genome plasticity are fairly quickly removed by purifying selection, but thatin the first 40 amino acids, for genes that need to be expressed at low levels, purifying selectionagainst slow codons is reduced, because poor translation is actually advantageous for thesegenes. To expand on this a bit, we feel that the 5000 or so proteins of the proteome have to beexpressed in the correct stoichiometric ratios, and that poor translation can be a useful tool tohelp achieve this. In this explanation, slow translation at the 5’ end is bad for translation (inagreement with our reporter experiments), but can be good for the organism, when it occurs infront of a gene that needs to be expressed poorly. Whereas, in Tuller, slow translation at the 5’end is good for translation.</p><p>Reviewer 2 wondered whether the N-terminal fusion peptide affects GFP fluorescence in ourreporter. This specific reporter, with this N-terminus, has been characterized by Dean andGrayhack (2012), and by Gamble et al. (2016), and the idea that a super-folder GFP reporter isnot greatly affected by N-terminal fusions is based on the work of Pedelacq (2006). None ofthese papers show whether this N-terminal fusion might have some effect, but together, theyprovide good reason to think that any effect would be small. These citations have been added.</p></body></sub-article></article>