<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Archiving and Interchange DTD v1.2 20190208//EN"  "JATS-archivearticle1.dtd"><article article-type="research-article" dtd-version="1.2" xmlns:ali="http://www.niso.org/schemas/ali/1.0/" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink"><front><journal-meta><journal-id journal-id-type="nlm-ta">elife</journal-id><journal-id journal-id-type="publisher-id">eLife</journal-id><journal-title-group><journal-title>eLife</journal-title></journal-title-group><issn pub-type="epub" publication-format="electronic">2050-084X</issn><publisher><publisher-name>eLife Sciences Publications, Ltd</publisher-name></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">67806</article-id><article-id pub-id-type="doi">10.7554/eLife.67806</article-id><article-categories><subj-group subj-group-type="display-channel"><subject>Research Article</subject></subj-group><subj-group subj-group-type="heading"><subject>Evolutionary Biology</subject></subj-group><subj-group subj-group-type="heading"><subject>Genetics and Genomics</subject></subj-group></article-categories><title-group><article-title>Mutational sources of <italic>trans</italic>-regulatory variation affecting gene expression in <italic>Saccharomyces cerevisiae</italic></article-title></title-group><contrib-group><contrib contrib-type="author" corresp="yes" id="author-227774"><name><surname>Duveau</surname><given-names>Fabien</given-names></name><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0003-4784-0640</contrib-id><email>fabien.duveau@ens-lyon.fr</email><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="other" rid="fund3"/><xref ref-type="fn" rid="con1"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author" id="author-229841"><name><surname>Vande Zande</surname><given-names>Petra</given-names></name><xref ref-type="aff" rid="aff3">3</xref><xref ref-type="other" rid="fund5"/><xref ref-type="fn" rid="con2"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author" id="author-110238"><name><surname>Metzger</surname><given-names>Brian PH</given-names></name><contrib-id authenticated="true" contrib-id-type="orcid">http://orcid.org/0000-0003-4878-2913</contrib-id><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="other" rid="fund6"/><xref ref-type="fn" rid="con3"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author" id="author-229842"><name><surname>Diaz</surname><given-names>Crisandra J</given-names></name><xref ref-type="aff" rid="aff3">3</xref><xref ref-type="fn" rid="con4"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author" id="author-110241"><name><surname>Walker</surname><given-names>Elizabeth A</given-names></name><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="fn" rid="con5"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author" id="author-229843"><name><surname>Tryban</surname><given-names>Stephen</given-names></name><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="fn" rid="con6"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author" id="author-229844"><name><surname>Siddiq</surname><given-names>Mohammad A</given-names></name><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="other" rid="fund7"/><xref ref-type="other" rid="fund8"/><xref ref-type="fn" rid="con7"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author" id="author-110239"><name><surname>Yang</surname><given-names>Bing</given-names></name><xref ref-type="aff" rid="aff3">3</xref><xref ref-type="fn" rid="con8"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author" corresp="yes" id="author-42157"><name><surname>Wittkopp</surname><given-names>Patricia J</given-names></name><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0001-7619-0048</contrib-id><email>wittkopp@umich.edu</email><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff3">3</xref><xref ref-type="other" rid="fund1"/><xref ref-type="other" rid="fund2"/><xref ref-type="other" rid="fund4"/><xref ref-type="fn" rid="con9"/><xref ref-type="fn" rid="conf2"/></contrib><aff id="aff1"><label>1</label><institution>Department of Ecology and Evolutionary Biology, University of Michigan</institution><addr-line><named-content content-type="city">Ann Arbor</named-content></addr-line><country>United States</country></aff><aff id="aff2"><label>2</label><institution>Laboratory of Biology and Modeling of the Cell, Ecole Normale Supérieure de Lyon, CNRS, Université Claude Bernard Lyon, Université de Lyon</institution><addr-line><named-content content-type="city">Lyon</named-content></addr-line><country>France</country></aff><aff id="aff3"><label>3</label><institution>Department of Molecular, Cellular, and Developmental Biology, University of Michigan</institution><addr-line><named-content content-type="city">Ann Arbor</named-content></addr-line><country>United States</country></aff></contrib-group><contrib-group content-type="section"><contrib contrib-type="editor"><name><surname>Landry</surname><given-names>Christian R</given-names></name><role>Reviewing Editor</role><aff><institution>Université Laval</institution><country>Canada</country></aff></contrib><contrib contrib-type="senior_editor"><name><surname>Przeworski</surname><given-names>Molly</given-names></name><role>Senior Editor</role><aff><institution>Columbia University</institution><country>United States</country></aff></contrib></contrib-group><pub-date date-type="publication" publication-format="electronic"><day>31</day><month>08</month><year>2021</year></pub-date><pub-date pub-type="collection"><year>2021</year></pub-date><volume>10</volume><elocation-id>e67806</elocation-id><history><date date-type="received" iso-8601-date="2021-02-23"><day>23</day><month>02</month><year>2021</year></date><date date-type="accepted" iso-8601-date="2021-08-03"><day>03</day><month>08</month><year>2021</year></date></history><pub-history><event><event-desc>This manuscript was published as a preprint at bioRxiv.</event-desc><date date-type="preprint" iso-8601-date="2021-02-22"><day>22</day><month>02</month><year>2021</year></date><self-uri content-type="preprint" xlink:href="https://doi.org/10.1101/2021.02.22.432283"/></event></pub-history><permissions><copyright-statement>© 2021, Duveau et al</copyright-statement><copyright-year>2021</copyright-year><copyright-holder>Duveau et al</copyright-holder><ali:free_to_read/><license xlink:href="http://creativecommons.org/licenses/by/4.0/"><ali:license_ref>http://creativecommons.org/licenses/by/4.0/</ali:license_ref><license-p>This article is distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="http://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution License</ext-link>, which permits unrestricted use and redistribution provided that the original author and source are credited.</license-p></license></permissions><self-uri content-type="pdf" xlink:href="elife-67806-v1.pdf"/><abstract><p>Heritable variation in a gene’s expression arises from mutations impacting <italic>cis</italic>- and <italic>trans</italic>-acting components of its regulatory network. Here, we investigate how <italic>trans</italic>-regulatory mutations are distributed within the genome and within a gene regulatory network by identifying and characterizing 69 mutations with <italic>trans</italic>-regulatory effects on expression of the same focal gene in <italic>Saccharomyces cerevisiae</italic>. Relative to 1766 mutations without effects on expression of this focal gene, we found that these <italic>trans</italic>-regulatory mutations were enriched in coding sequences of transcription factors previously predicted to regulate expression of the focal gene. However, over 90% of the <italic>trans</italic>-regulatory mutations identified mapped to other types of genes involved in diverse biological processes including chromatin state, metabolism, and signal transduction. These data show how genetic changes in diverse types of genes can impact a gene’s expression in <italic>trans</italic>, revealing properties of <italic>trans</italic>-regulatory mutations that provide the raw material for <italic>trans</italic>-regulatory variation segregating within natural populations.</p></abstract><kwd-group kwd-group-type="author-keywords"><kwd>gene regulation</kwd><kwd>bulk segregant analysis</kwd><kwd>TDH3</kwd><kwd>RAP1</kwd><kwd>GCR1</kwd><kwd>eQTL</kwd></kwd-group><kwd-group kwd-group-type="research-organism"><title>Research organism</title><kwd><italic>S. cerevisiae</italic></kwd></kwd-group><funding-group><award-group id="fund1"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100000002</institution-id><institution>National Institutes of Health</institution></institution-wrap></funding-source><award-id>R01GM108826</award-id><principal-award-recipient><name><surname>Wittkopp</surname><given-names>Patricia J</given-names></name></principal-award-recipient></award-group><award-group id="fund2"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100000002</institution-id><institution>National Institutes of Health</institution></institution-wrap></funding-source><award-id>R35GM118073</award-id><principal-award-recipient><name><surname>Wittkopp</surname><given-names>Patricia J</given-names></name></principal-award-recipient></award-group><award-group id="fund3"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100004410</institution-id><institution>European Molecular Biology Organization</institution></institution-wrap></funding-source><award-id>1114-2012</award-id><principal-award-recipient><name><surname>Duveau</surname><given-names>Fabien</given-names></name></principal-award-recipient></award-group><award-group id="fund4"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100000001</institution-id><institution>National Science Foundation</institution></institution-wrap></funding-source><award-id>MCB-1929737</award-id><principal-award-recipient><name><surname>Wittkopp</surname><given-names>Patricia J</given-names></name></principal-award-recipient></award-group><award-group id="fund5"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100000002</institution-id><institution>National Institutes of Health</institution></institution-wrap></funding-source><award-id>T32GM007544</award-id><principal-award-recipient><name><surname>Vande Zande</surname><given-names>Petra</given-names></name></principal-award-recipient></award-group><award-group id="fund6"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100000002</institution-id><institution>National Institutes of Health</institution></institution-wrap></funding-source><award-id>T32HG000040</award-id><principal-award-recipient><name><surname>Metzger</surname><given-names>Brian PH</given-names></name></principal-award-recipient></award-group><award-group id="fund7"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100000002</institution-id><institution>National Institutes of Health</institution></institution-wrap></funding-source><award-id>T32HG000040</award-id><principal-award-recipient><name><surname>Siddiq</surname><given-names>Mohammad A</given-names></name></principal-award-recipient></award-group><award-group id="fund8"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100007270</institution-id><institution>University of Michigan</institution></institution-wrap></funding-source><award-id>Michigan Life Sciences Fellow program</award-id><principal-award-recipient><name><surname>Siddiq</surname><given-names>Mohammad A</given-names></name></principal-award-recipient></award-group><funding-statement>The funders had no role in study design, data collection and interpretation, or the decision to submit the work for publication.</funding-statement></funding-group><custom-meta-group><custom-meta specific-use="meta-only"><meta-name>Author impact statement</meta-name><meta-value>Mapping mutations affecting gene expression within the yeast genome and within a gene regulatory network reveals properties of the raw material for regulatory variation.</meta-value></custom-meta></custom-meta-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>The regulation of gene expression is a complex process, essential for cellular function, that impacts development, physiology, and evolution. Expression of each gene is regulated by its <italic>cis</italic>-regulatory DNA sequences (e.g. promoters, enhancers) interacting either directly or indirectly with <italic>trans</italic>-acting factors (e.g. transcription factors, signaling pathways) encoded by genes throughout the genome. Genetic variants affecting both <italic>cis-</italic> and <italic>trans</italic>-acting components of regulatory networks contribute to expression differences within and between species (<xref ref-type="bibr" rid="bib2">Albert and Kruglyak, 2015</xref>; <xref ref-type="bibr" rid="bib3">Barbeira et al., 2018</xref>; <xref ref-type="bibr" rid="bib13">Ferraro et al., 2020</xref>; <xref ref-type="bibr" rid="bib15">Gamazon et al., 2018</xref>; <xref ref-type="bibr" rid="bib58">Oliver et al., 2005</xref>). This regulatory variation arises the same way as genetic variation affecting any other quantitative trait: new mutations generate variation in gene expression and selection favors the transmission of some genetic variants over others, giving rise to polymorphism within a species and divergence between species. Because new mutations are the raw material for this polymorphism and divergence, knowing how new mutations impact gene expression is essential for understanding how gene regulation evolves (reviewed in <xref ref-type="bibr" rid="bib22">Hill et al., 2021</xref>). Targeted mutagenesis has been used to systematically examine the effects of individual mutations in <italic>cis</italic>-regulatory sequences for a variety of elements in a variety of species (<xref ref-type="bibr" rid="bib24">Hornung et al., 2012</xref>; <xref ref-type="bibr" rid="bib36">Kwasnieski et al., 2012</xref>; <xref ref-type="bibr" rid="bib46">Maricque et al., 2017</xref>; <xref ref-type="bibr" rid="bib51">Melnikov et al., 2012</xref>; <xref ref-type="bibr" rid="bib52">Metzger et al., 2015</xref>; <xref ref-type="bibr" rid="bib61">Patwardhan et al., 2009</xref>; <xref ref-type="bibr" rid="bib69">Sharon et al., 2012</xref>), but such targeted approaches are not well-suited for surveying the effects of new <italic>trans</italic>-regulatory mutations because <italic>trans</italic>-regulatory mutations can be located virtually anywhere within the genome. Consequently, we know comparatively little about the genomic sources, molecular mechanisms of action and evolutionary contributions of individual <italic>trans</italic>-regulatory mutations.</p><p>Genetic mapping experiments and genome-wide association studies (GWAS) have shown that gene expression is a highly polygenic trait, with hundreds of genetic variants typically associated with natural variation in expression levels of each gene (<xref ref-type="bibr" rid="bib1">Albert et al., 2018</xref>; <xref ref-type="bibr" rid="bib54">Metzger and Wittkopp, 2019</xref>; <xref ref-type="bibr" rid="bib72">Sinnott-Armstrong et al., 2021</xref>). Although these studies often lack the resolution to identify individual genetic changes affecting expression, most of this variation maps far from the gene whose expression it affects and is therefore likely to have <italic>trans</italic>-acting effects. <italic>Trans</italic>-acting variants segregating in natural populations are most often expected to affect transcription factors (<xref ref-type="bibr" rid="bib1">Albert et al., 2018</xref>; <xref ref-type="bibr" rid="bib39">Lewis et al., 2014</xref>), but they can also alter genes encoding signaling proteins, chromatin modifiers, metabolic enzymes, or any other gene product that can influence the availability, accessibility, or activity of transcription factors (<xref ref-type="bibr" rid="bib45">Lutz et al., 2019</xref>; <xref ref-type="bibr" rid="bib50">Mehrabian et al., 2005</xref>; <xref ref-type="bibr" rid="bib68">Schadt et al., 2005</xref>; <xref ref-type="bibr" rid="bib86">Yvert et al., 2003</xref>). Indeed, the recently proposed omnigenic model emphasizes the interconnectedness of regulatory networks controlling transcription to help explain the highly polygenic nature of diverse quantitative traits.</p><p>Despite the vast potential target size for <italic>trans</italic>-regulatory mutations (<xref ref-type="bibr" rid="bib22">Hill et al., 2021</xref>), regions of the genome most likely to harbor mutations affecting a particular gene’s expression might be predictable from knowledge of its regulatory network. Among eukaryotes, the set of genes and interactions regulating gene expression in <italic>trans</italic> is perhaps best understood in the baker’s yeast <italic>Saccharomyces cerevisiae</italic> (<xref ref-type="bibr" rid="bib27">Hughes and de Boer, 2013</xref>): networks of regulatory connections (<xref ref-type="bibr" rid="bib76">Teixeira et al., 2018</xref>) have been inferred from experiments that profile the transcriptional effects of gene deletions (<xref ref-type="bibr" rid="bib26">Hughes et al., 2000</xref>; <xref ref-type="bibr" rid="bib29">Jackson et al., 2020</xref>; <xref ref-type="bibr" rid="bib32">Kemmeren et al., 2014</xref>), map binding sites for transcription factors (<xref ref-type="bibr" rid="bib64">Rhee and Pugh, 2011</xref>; <xref ref-type="bibr" rid="bib87">Zheng et al., 2010</xref>; <xref ref-type="bibr" rid="bib88">Zhu et al., 2009</xref>), identify protein-protein interactions (<xref ref-type="bibr" rid="bib17">Gavin et al., 2002</xref>; <xref ref-type="bibr" rid="bib42">Liu et al., 2020</xref>; <xref ref-type="bibr" rid="bib75">Tarassov et al., 2008</xref>), and test pairs of genes for genetic interactions (<xref ref-type="bibr" rid="bib7">Costanzo et al., 2016</xref>; <xref ref-type="bibr" rid="bib80">van Leeuwen et al., 2016</xref>). However, the extent to which the genomic sources of <italic>trans</italic>-regulatory mutations can be predicted from such networks is generally unknown (<xref ref-type="bibr" rid="bib14">Flint and Ideker, 2019</xref>). In addition, the extent to which the genomic distribution of new mutations predicts the genomic distribution of natural polymorphisms is also unclear because mutations that are strongly deleterious might rarely be found circulating within a population as standing genetic variation. For example, mutations in coding sequences might often impact gene expression but might also tend to be more pleiotropic and thus more deleterious than mutations in non-coding regions of these genes. Comparing the genomic distribution of mutations that have not experienced natural selection to the genomic distribution of polymorphisms that have can reveal such differences between the possible and actual sources of variation in gene expression in the wild.</p><p>Systematic studies of new mutations identifying and characterizing the effects of individual genetic changes are thus an important complement to GWAS describing the polygenic variation segregating within a species. Recently, a chemical mutagen was used to induce mutations throughout the genome of <italic>S. cerevisiae,</italic> and hundreds of mutant genotypes were collected that all altered expression of the same gene, providing the biological resources needed to systematically characterize properties of new <italic>trans</italic>-regulatory mutations and to test the predictive power of inferred regulatory networks (<xref ref-type="bibr" rid="bib20">Gruber et al., 2012</xref>; <xref ref-type="bibr" rid="bib53">Metzger et al., 2016</xref>). Here, we use genetic mapping, candidate gene sequencing and functional validation to identify 69 <italic>trans</italic>-regulatory mutations that alter expression of the focal gene from this set of mutants and contrast their properties with a comparable set of 1766 mutations that did not affect expression of the focal gene.</p><p>Using this collection of individual <italic>trans</italic>-regulatory mutations, we determined how <italic>trans</italic>-regulatory mutations affecting expression of a single gene were distributed within the genome and within a regulatory network. For example, we asked how frequently <italic>trans</italic>-regulatory mutations were located in coding or non-coding sequences because <italic>trans</italic>-regulatory variants are often predicted to affect coding sequences (<xref ref-type="bibr" rid="bib22">Hill et al., 2021</xref>) but some non-coding variants have been shown to be associated with <italic>trans</italic>-regulatory effects on gene expression (<xref ref-type="bibr" rid="bib21">GTEx Consortium, 2020</xref>; <xref ref-type="bibr" rid="bib85">Yao et al., 2017</xref>; <xref ref-type="bibr" rid="bib86">Yvert et al., 2003</xref>). We also asked whether genes encoding transcription factors were the primary source of <italic>trans-</italic>regulatory variation, which is often assumed (<xref ref-type="bibr" rid="bib1">Albert et al., 2018</xref>; <xref ref-type="bibr" rid="bib39">Lewis et al., 2014</xref>) despite case studies identifying <italic>trans</italic>-regulatory variants in genes encoding proteins with other functions (<xref ref-type="bibr" rid="bib45">Lutz et al., 2019</xref>; <xref ref-type="bibr" rid="bib50">Mehrabian et al., 2005</xref>; <xref ref-type="bibr" rid="bib68">Schadt et al., 2005</xref>; <xref ref-type="bibr" rid="bib86">Yvert et al., 2003</xref>). To determine how well an inferred regulatory network can predict genomic sources of expression changes, we mapped the <italic>trans</italic>-regulatory mutations to a network of transcription factors predicted by functional genomic data to regulate expression of the focal gene and examined the molecular functions and biological processes impacted by <italic>trans</italic>-regulatory mutations that did not map to genes in this network. By systematically examining the properties and identity of new <italic>trans</italic>-regulatory mutations, this work fills a key gap in our understanding of how expression differences arise and may help predict sources of <italic>trans</italic>-regulatory variation segregating in natural populations. Indeed, we found that the genomic distribution of new <italic>trans</italic>-regulatory mutations overlaps significantly with the genomic distribution of <italic>trans</italic>-regulatory variants segregating among wild isolates of <italic>S. cerevisiae</italic> that affect expression of the same gene (<xref ref-type="bibr" rid="bib54">Metzger and Wittkopp, 2019</xref>), suggesting that the mutational process generating new <italic>trans</italic>-regulatory variation significantly shaped the regulatory variation we see in the wild.</p></sec><sec id="s2" sec-type="results"><title>Results</title><sec id="s2-1"><title>Genetic mapping of <italic>trans</italic>-regulatory mutations</title><p>To characterize properties of new <italic>trans</italic>-regulatory mutations affecting expression of a focal gene, we took advantage of three previously collected sets of haploid mutants that all showed altered expression of the same reporter gene (<xref ref-type="fig" rid="fig1">Figure 1A</xref>, <xref ref-type="bibr" rid="bib20">Gruber et al., 2012</xref>; <xref ref-type="bibr" rid="bib53">Metzger et al., 2016</xref>). This reporter gene (<italic>P<sub>TDH3</sub>-YFP</italic>) encodes a yellow fluorescent protein whose expression is regulated by the <italic>S. cerevisiae TDH3</italic> promoter, which natively drives constitutive expression of a glyceraldehyde-3-phosphate dehydrogenase involved in glycolysis and gluconeogenesis (<xref ref-type="bibr" rid="bib48">McAlister and Holland, 1985</xref>). The mutation rate was increased to obtain these mutants by exposure to the chemical mutagen ethyl methanesulfonate (EMS), which induces primarily G:C to A:T point mutations randomly throughout the genome (<xref ref-type="bibr" rid="bib71">Shiwa et al., 2012</xref>). The dose of EMS used in these studies was chosen so that most mutants with a detectable change in <italic>P<sub>TDH3</sub>-YFP</italic> expression should have only one mutation causing this change in expression among the mutations they carry (<xref ref-type="bibr" rid="bib53">Metzger et al., 2016</xref>; <xref ref-type="bibr" rid="bib20">Gruber et al., 2012</xref>). Together, these collections contain ~1500 mutants isolated irrespective of their fluorescence levels (‘unenriched’ mutants) and ~1200 mutants isolated after enriching for cells with the largest changes in fluorescence (<xref ref-type="fig" rid="fig1">Figure 1A</xref>, see <xref ref-type="fig" rid="fig1s1">Figure 1—figure supplement 1</xref> for a diagram showing the number of mutants and mutations included at each step of the study). When we started this work, expression level of <italic>P<sub>TDH3</sub>-YFP</italic> in these mutant genotypes had been described (<xref ref-type="bibr" rid="bib20">Gruber et al., 2012</xref>; <xref ref-type="bibr" rid="bib53">Metzger et al., 2016</xref>), but the specific mutations present within each mutant as well as which mutation(s) alter(s) <italic>P<sub>TDH3</sub>-YFP</italic> expression in each genotype were unknown.</p><fig-group><fig id="fig1" position="float"><label>Figure 1.</label><caption><title>Mutant strains analyzed with altered expression of a <italic>P<sub>TDH3</sub>-YFP</italic> reporter gene.</title><p>(<bold>A</bold>) Summary of the three previously published collections of <italic>S. cerevisiae</italic> mutants obtained by ethyl methanesulfonate (EMS) mutagenesis of a haploid strain expressing a yellow fluorescent protein (YFP) under control of the <italic>TDH3</italic> promoter. *One mutant is included in both columns because it was analyzed both by BSA-Seq and Sanger sequencing. (<bold>B–D</bold>) Previously published fluorescence levels (x-axis) and statistical significance of the difference in median fluorescence between each mutant and the un-mutagenized progenitor strain (y-axis) are shown for mutants analyzed in (<bold>B</bold>) <xref ref-type="bibr" rid="bib20">Gruber et al., 2012</xref> and (<bold>C,D</bold>) <xref ref-type="bibr" rid="bib53">Metzger et al., 2016</xref>. (<bold>B</bold>) Collection of 1064 mutants from <xref ref-type="bibr" rid="bib20">Gruber et al., 2012</xref> enriched for mutations causing large fluorescence changes. p-values were computed using <italic>Z</italic>-tests in this study, based on one measure of fluorescence for each mutant and 30 measures of fluorescence for the progenitor strain. (<bold>C</bold>) Collection of 211 mutants from <xref ref-type="bibr" rid="bib53">Metzger et al., 2016</xref> enriched for mutations causing large fluorescence changes. (<bold>D</bold>) Collection of 1498 mutants from <xref ref-type="bibr" rid="bib53">Metzger et al., 2016</xref> obtained irrespective of their fluorescence levels (unenriched mutants). (<bold>E</bold>) A new fluorescence dataset for 197 unenriched mutants from <xref ref-type="bibr" rid="bib53">Metzger et al., 2016</xref> (blue in panel D) that were reanalyzed in a 2nd screen as part of this study. (<bold>C–E</bold>) Four replicate populations were analyzed for each mutant. Error bars show 95% confidence intervals of fluorescence levels measured among these replicates. p-values were obtained using the permutation tests described in Methods. (<bold>B–E</bold>) Mutants analyzed by BSA-Seq are highlighted in red. All of these mutants showed fluorescence changes greater than 0.01 (vertical dotted lines) and p-value below 0.05 (horizontal dotted lines); percentages of all mutants that met these selection criteria in each collection are also shown. Mutants selected for Sanger sequencing of the <italic>ADE4</italic>, <italic>ADE5</italic>, and/or <italic>ADE6</italic> candidate genes are highlighted in green. The mutant analyzed with both BSA-seq and Sanger sequencing is both red and green in panel (<bold>C</bold>). Two mutants selected for Sanger sequencing of the <italic>ADE2</italic> gene are highlighted in purple, one in (<bold>D</bold>) and one in (<bold>E</bold>).</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-67806-fig1-v1.tif"/></fig><fig id="fig1s1" position="float" specific-use="child-fig"><label>Figure 1—figure supplement 1.</label><caption><title>Diagram showing the number of mutant strains and mutations considered at each step of the study.</title></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-67806-fig1-figsupp1-v1.tif"/></fig></fig-group><p>From these collections, we selected 82 EMS-treated mutants for genetic mapping to identify individual causal mutations (<xref ref-type="fig" rid="fig1">Figure 1A</xref>, <xref ref-type="fig" rid="fig1s1">Figure 1—figure supplement 1</xref>). Sanger sequencing of the reporter gene in these mutants showed that none had mutations in the <italic>TDH3</italic> promoter or any other part of the reporter gene, indicating that they harbored mutations affecting <italic>P<sub>TDH3</sub>-YFP</italic> expression in <italic>trans</italic>. Thirty-nine of these mutants were selected based on previously published fluorescence data, with 11 mutants selected from the collections enriched for large effects (red points in <xref ref-type="fig" rid="fig1">Figure 1B,C</xref>) and 28 mutants selected from the unenriched collection (red points in <xref ref-type="fig" rid="fig1">Figure 1D</xref>). Each selected mutant showed changes in average YFP fluorescence greater than 1% relative to the un-mutagenized progenitor strain. Another 197 mutants from the unenriched collection (blue points in <xref ref-type="fig" rid="fig1">Figure 1D</xref>) were subjected to a secondary fluorescence screen, from which an additional 43 mutants with a change in fluorescence greater than 1% (red points in <xref ref-type="fig" rid="fig1">Figure 1E</xref>) were chosen. Overall, the 82 mutants were selected randomly from the 528 EMS mutants that showed statistically significant fluorescence changes greater than 1% relative to wild-type (p &lt; 0.05, see Methods and <xref ref-type="fig" rid="fig1">Figure 1</xref> legend for a description of the statistical tests). A 1% change in YFP fluorescence has previously been shown to correspond to a ~3% change in YFP mRNA abundance (see <xref ref-type="bibr" rid="bib11">Duveau et al., 2018</xref>), although changes in fluorescence caused by <italic>trans</italic>-regulatory mutations in these mutants could affect either transcription driven by the <italic>TDH3</italic> promoter or post-transcriptional regulation of YFP synthesis or stability.</p><p>To identify mutations within the 82 selected EMS mutants, and to determine which of these mutation(s) were most likely to affect YFP expression in each mutant, we performed bulk-segregant analysis followed by whole-genome sequencing (BSA-Seq) as described in <xref ref-type="bibr" rid="bib10">Duveau et al., 2014</xref> with minor modifications (see Methods). Briefly, each mutant strain was crossed to a common mapping strain expressing the <italic>P<sub>TDH3</sub>-YFP</italic> reporter gene, and large populations of random haploid spores were isolated after inducing meiosis in the resulting diploids (<xref ref-type="fig" rid="fig2">Figure 2A</xref>). For each of the 82 segregant populations, a low fluorescent bulk and a high fluorescent bulk of ~1.5 x 10<sup>5</sup> cells each were isolated using fluorescence-activated cell sorting (FACS) (<xref ref-type="fig" rid="fig2">Figure 2B</xref>). Genomic DNA extracted from each bulk was then sequenced to an average coverage of ~105x (ranging from 75x to 134x among samples, <xref ref-type="supplementary-material" rid="supp1">Supplementary file 1</xref>) to identify the mutations present within each mutant genotype and to quantify the frequency of mutant and non-mutant alleles in both bulks (<xref ref-type="fig" rid="fig2">Figure 2C</xref>). A mutation causing a change in fluorescence is expected to be found at different frequencies in the two populations of segregant cells. Conversely, a mutation with no effect on fluorescence that is not genetically linked to a mutation affecting fluorescence is expected to be found at similar frequencies in these two populations.</p><fig-group><fig id="fig2" position="float"><label>Figure 2.</label><caption><title>Genetic mapping and functional testing of <italic>trans</italic>-regulatory mutations affecting <italic>P<sub>TDH3</sub>-YFP</italic> expression.</title><p>(<bold>A–C</bold>) Overview of the BSA-Seq approach. (<bold>A</bold>) Crossing scheme used to map mutations in each EMS mutant strain by crossing to an un-mutagenized strain expressing <italic>P<sub>TDH3</sub>-YFP</italic>. Stars indicate hypothetical mutations. (<bold>B</bold>) Isolation of two bulks of haploid segregants with high and low fluorescence levels (see Methods). (<bold>C</bold>) Estimation of allele frequencies in each bulk using high-throughput sequencing. A mutation without effect on fluorescence is found at similar frequencies in the two bulks (white stars). A mutation affecting fluorescence or genetically linked to a mutation affecting fluorescence is found at different frequencies between the two bulks (red stars). (<bold>D</bold>) Type of mutations identified in BSA-Seq data for the 76 mutants from <xref ref-type="bibr" rid="bib53">Metzger et al., 2016</xref>. (<bold>E</bold>) Median expression of <italic>P<sub>TDH3</sub>-YFP</italic> is shown for the wild-type (WT) progenitor strain (black), for five EMS mutants (brown) with two linked mutations associated with fluorescence in BSA-Seq data and for 10 single-site mutants (turquoise) carrying one of the two linked mutations in the five EMS mutants. Single-site mutants are grouped in pairs next to the EMS mutant carrying the same mutations and are named after the gene that they affect. Expression levels are expressed relative to the wild-type progenitor strain. For each strain, dots represent the median expression measured for each replicate population and tick marks represent the mean of median expression from replicate populations. (<bold>F</bold>) Effects of mutations associated with fluorescence in BSA-Seq experiments tested in single-site mutants. X-axis: Effect of each mutation on expression measured in a single site mutant and relative to the wild-type progenitor strain. Error bars are 95% confidence intervals obtained from at least four replicate populations. Y-axis: <italic>G</italic> statistics of the tests used to compare the frequencies of each mutation between the two bulks in BSA-Seq experiments, with a negative sign if the mutation was more frequent in the low fluorescence bulk and a positive sign if the mutation was more frequent in the high fluorescence bulk. One single-site mutant (<italic>NAP1</italic>, red) showed no significant change in expression relative to the wild-type progenitor strain (<italic>t</italic>-test, p-value &gt; 0.05); the mutation it carries is therefore considered to be a false positive in the BSA-seq data. For two other single-site mutants (<italic>ATP23</italic> and <italic>IRA2</italic>, green), the expression changes were not in the same direction as predicted by the signed <italic>G</italic>-values. (<bold>G</bold>) <italic>P<sub>TDH3</sub>-YFP</italic> expression levels in single-site mutants and in EMS mutants sharing the same mutation. Data points represent median expression levels of 40 EMS mutants (x-axis) and 40 single-site mutants (y-axis) measured by flow cytometry in four replicate populations. Circles: mutations identified by BSA-Seq. Triangles: mutations identified by sequencing candidate genes. Error bars: 95% confidence intervals of expression levels obtained from replicate populations. Data points are colored based on the p-values of permutation tests used to assess the statistical significance of expression differences between each single site mutant and the EMS mutant carrying the same mutation (see <xref ref-type="fig" rid="fig2s5">Figure 2—figure supplement 5</xref> for details). The light blue area represents the 95% confidence interval of expression differences between genetically identical samples across the whole range of median expression values. This confidence interval was calculated from a null distribution described in <xref ref-type="fig" rid="fig2s5">Figure 2—figure supplement 5A</xref>. (<bold>E–G</bold>) Expression levels are expressed on a scale linearly related to <italic>YFP</italic> mRNA levels and relative to the median expression of the wild-type progenitor strain (see Materials and methods).</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-67806-fig2-v1.tif"/></fig><fig id="fig2s1" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 1.</label><caption><title>Number of mutations per strain identified from BSA-Seq data.</title><p>Data from 76 EMS mutants from <xref ref-type="bibr" rid="bib53">Metzger et al., 2016</xref> are shown. Vertical dotted line: mean number of mutations per strain (23.9). Blue dots and line: Poisson distribution with λ = 23.9 and k = 76 representing the expected numbers of mutations per line if mutations had the same probability of occurring in all mutant lines.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-67806-fig2-figsupp1-v1.tif"/></fig><fig id="fig2s2" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 2.</label><caption><title>Magnitude of expression changes in EMS mutants depending on the number of mutations associated with fluorescence in BSA-Seq experiments.</title><p>Individual data points represent absolute differences between the median expression levels of EMS mutants and of the un-mutagenized progenitor strain averaged among four replicate populations. Mutations that were associated with fluorescence only because of genetic linkage (i.e. without additional evidence of affecting expression) were not counted (see <xref ref-type="supplementary-material" rid="supp4">Supplementary file 4</xref>). Blue dots: mutants with decreased expression relative to the progenitor strain. Red dots: mutants with increased expression relative to the progenitor strain. Using Mann-Whitney-Wilcoxon tests, the magnitude of expression changes was found to be significantly lower for mutants without any mutation associated with fluorescence than for mutants with 1 (p = 5.3 x 10<sup>−5</sup>) or 2 (p = 0.018) mutations associated with fluorescence.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-67806-fig2-figsupp2-v1.tif"/></fig><fig id="fig2s3" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 3.</label><caption><title>Relationship between the number of mutations per EMS mutant strain and the absolute expression change relative to the progenitor strain.</title><p>This relationship is shown for EMS mutants without any mutation associated with fluorescence in BSA-Seq data (green dots and green regression line) as well as for EMS mutants with at least one mutation associated with fluorescence in BSA-Seq data (gray dots and gray regression line). Mutations that were associated with fluorescence only because of genetic linkage and without other evidence of affecting expression were excluded (see <xref ref-type="supplementary-material" rid="supp4">Supplementary file 4</xref>). <italic>F</italic>-tests were used to assess the statistical significance of linear regressions. A significant relationship was observed between the number of mutations per mutant strain and the absolute expression change only when no mutation was associated with fluorescence (r<sup>2</sup> = 0.127, p-value = 0.03). This observation supports the hypothesis that several mutations with small effects could collectively contribute to the expression change observed in mutants for which no mutation was associated with fluorescence. The small effects of these mutations would explain why they were not associated with fluorescence in the BSA-Seq analyses.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-67806-fig2-figsupp3-v1.tif"/></fig><fig id="fig2s4" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 4.</label><caption><title>Effects of individual mutations in purine biosynthesis genes on YFP expression levels differ among promoters.</title><p>Each dot indicates the median fluorescence level of at least 5 x 10<sup>4</sup> cells for each genotype averaged across three experimental replicates. Error bars represent median absolute deviation across replicates. Dots are grouped along the x-axis based on the yeast promoter used to drive YFP expression (<italic>P<sub>GPD1</sub></italic>, <italic>P<sub>RNR1</sub></italic>, <italic>P<sub>STM1</sub></italic>, and <italic>P<sub>TDH3</sub></italic>), with ‘None’ corresponding to the autofluorescence measured in a strain without a fluorescent reporter gene. The color of each dot indicates which mutation was introduced in one of the genes involved in de novo purine synthesis (<italic>ADE2</italic>, <italic>ADE5</italic> or <italic>ADE6</italic>), with the specific mutation introduced indicated in the key. The goal of this experiment was to determine whether the regulatory mutations identified in purine synthesis genes altered <italic>P<sub>TDH3</sub>-YFP</italic> expression at the transcriptional or post-transcriptional level. If the mutations acted post-transcriptionally, their effect on fluorescence level should be the same among strains with different promoters driving YFP expression because they all produce the same <italic>YFP</italic> transcript. However, we observed that the mutations in purine synthesis genes increased fluorescence level when YFP expression was driven by the <italic>TDH3</italic> or the <italic>GPD1</italic> promoter but not when YFP expression was driven by the <italic>RNR1</italic> or the <italic>STM1</italic> promoter, indicating that the effects of these mutations on YFP expression were promoter specific.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-67806-fig2-figsupp4-v1.tif"/></fig><fig id="fig2s5" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 5.</label><caption><title>Factors contributing to expression differences observed between EMS and single-site mutants.</title><p>(<bold>A</bold>) Distribution of absolute expression differences observed between EMS and single-site mutants (bars). To assess the statistical significance of these expression differences, we estimated the magnitude of expression differences expected to arise by chance between genetically identical strains grown at different positions of a 96-well plate (red line). This null distribution was obtained from the differences in expression measured for 10,440 pairs of the un-mutagenized progenitor strain grown at different well positions in four replicate populations. We next randomly permuted 10<sup>5</sup> times the expression values between (i) each pair of EMS and single-site mutants and (ii) random pairs of the progenitor strain to calculate the one-sided <italic>p</italic>-value for each pair of mutants (i.e. the proportion of randomized expression differences greater than the observed expression difference). After Benjamini-Hochberg correction for multiple testing, we found that the expression difference between the single-site mutant and the EMS mutant carrying the same mutation was statistically significant (adjusted p-value &lt; 0.05) for 14 out of the 40 pairs of mutants (35%, red and blue bars), but highly significant (adjusted p-value &lt; 0.01) for only one pair (2.5%, red bar). Because mutant strains were exposed to the same micro-environmental and technical variation as the control samples used to establish the null distribution, these sources of variation are unlikely to explain the significant differences of expression observed between EMS and single-site mutants. Panels (<bold>B–F</bold>) test three other hypotheses to explain expression differences observed between single-site and EMS mutants. (<bold>B</bold>) Hypothesis 1: expression differences between EMS and single-site mutants are explained by differences in expression noise (i.e. the variability of expression observed among genetically identical cells grown in the same environment) among mutants. To test this hypothesis, we compared the expression noise measured by flow cytometry for each EMS mutant (x-axis) to the absolute difference of median expression levels between this EMS mutant and the corresponding single-site mutant (y-axis). We observed no significant correlation between the two parameters (r = 0.06, p-value = 0.71), indicating that expression noise is unlikely to explain expression differences between EMS and single-site mutants. Expression noise was calculated for each sample as the standard deviation of expression among cells divided by the median expression and it is reported as the average value among four replicate populations relative to the expression noise of the wild-type progenitor strain. Dot colors: p-values as shown in panel A. Dot shapes: circles represent mutations identified by BSA-seq; triangles represent mutations identified by sequencing candidate genes. Error bars: 95% confidence intervals calculated from four replicate populations. (<bold>C–D</bold>) Hypothesis 2: expression differences between EMS and single-site mutants are explained by additional mutations present in the EMS mutants. (<bold>C</bold>) Testing effects of additional mutations associated with fluorescence: boxplot comparing the magnitude of expression differences when only one mutation was associated with fluorescence and when more than one mutation was associated with fluorescence in BSA-Seq experiments. The fact that no statistical difference was observed between the two classes (Mann-Whitney-Wilcoxon test, p = 0.192) suggests that expression differences between EMS and single-site mutants were not likely to be caused by additional mutations associated with fluorescence in the BSA-Seq data. (<bold>D</bold>) Testing effects of additional mutations with statistical support for an association with fluorescence below the significance threshold. Expression difference between EMS and single-site mutants (x-axis) was compared to the highest <italic>G</italic>-value that was below our significance threshold for considering a mutation to be associated with fluorescence in the BSA-Seq data from each mutant (y-axis). A significant correlation was observed between the two parameters (Pearson’s <italic>r</italic> = 0.48; p = 0.02), suggesting that some mutations with associations below our detection threshold in the BSA-Seq experiments might contribute to expression differences observed between EMS and single-site mutants. Dots represent individual pairs of EMS and single-site mutants sharing the same mutation (with random jitter). The red line represents the linear regression of the y-axis parameter on the x-axis parameter. (<bold>E–F</bold>) Hypothesis 3: expression differences between EMS and single-site mutants are explained by secondary mutation(s) or epigenetic changes that occurred during construction of single-site mutants. To test this hypothesis, we isolated two independent clones for 26 single-site mutants after transformation of the progenitor strain and measured the expression difference between the two clones. (<bold>E</bold>) A positive correlation was observed between the expression difference between EMS and single-site mutants (x-axis) and the expression difference between the two independent clones for each single-site mutant (y-axis). This positive correlation indicates that mutations with larger expression differences between the single-site and EMS mutants tended to also show larger expression differences between independent transformants. Dot colors: p-values as shown in panel A. (<bold>F</bold>) Boxplot also showed that the average magnitude of expression differences between independent clones was higher for single site mutants with a statistically significant expression difference between the single-site and EMS mutant sharing the same mutation (Mann-Whitney-Wilcoxon test, p = 0.008). Results from E and F suggest that secondary mutation(s) and/or epigenetic changes that unintentionally occurred in some of the single-site mutant clones likely contributed to expression differences between some EMS and single-site mutants. It is important to emphasize, however, that these expression differences were small in magnitude and that overall the expression level of single-site mutants was strongly correlated with the expression level of EMS mutants (<xref ref-type="fig" rid="fig3">Figure 3</xref>).</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-67806-fig2-figsupp5-v1.tif"/></fig></fig-group><p>Using a stringent approach for calling sequence variants (see Methods), we identified a total of 1819 mutations in the BSA-Seq data from the 76 mutants from <xref ref-type="bibr" rid="bib53">Metzger et al., 2016</xref>; <xref ref-type="supplementary-material" rid="supp2">Supplementary file 2</xref>, <xref ref-type="fig" rid="fig1s1">Figure 1—figure supplement 1</xref>, among which 1768 mutations (97.2%) were single nucleotide changes (<xref ref-type="fig" rid="fig2">Figure 2D</xref>). Of these single nucleotide changes, 96.3% were one of the two types of point mutations (G:C to A:T transitions) known to be primarily induced by EMS (<xref ref-type="bibr" rid="bib71">Shiwa et al., 2012</xref>). Forty-eight small indels and three aneuploidies, which could have arisen spontaneously or been introduced by EMS, were also identified. Of these three mutants with aneuploidies, two were found to have an extra copy of chromosome I and one was found to have an extra copy of chromosome V based on ~1.5-fold higher sequencing coverage of these chromosomes relative to the rest of the genome in the BSA-seq data from segregant populations (shown in <xref ref-type="supplementary-material" rid="supp3">Supplementary file 3</xref>). We identified an average of 23.9 mutations per strain, which is within the 95% confidence interval of 21–45 mutations per strain estimated previously from the frequency of canavanine resistant mutants (<xref ref-type="bibr" rid="bib53">Metzger et al., 2016</xref>). Surprisingly, the number of mutations per strain did not follow a Poisson distribution: we observed more strains with a number of mutations far from the average than expected for a Poisson process (p-value &lt; 10<sup>−5</sup>, resampling test; <xref ref-type="fig" rid="fig2s1">Figure 2—figure supplement 1</xref>), which could be explained by cell-to-cell heterogeneity in DNA repair after exposure to the mutagen (<xref ref-type="bibr" rid="bib41">Liu et al., 2019</xref>; <xref ref-type="bibr" rid="bib79">Uphoff et al., 2016</xref>).</p><p>At least one mutation was significantly associated with fluorescence in 46 of the mutants analyzed based on likelihood ratio tests (<italic>G</italic>-tests described in Materials and methods, <xref ref-type="supplementary-material" rid="supp2">Supplementary file 2</xref>), with a total of 67 mutations associated with fluorescence identified among these mutants (<xref ref-type="fig" rid="fig1s1">Figure 1—figure supplement 1</xref>), including all three aneuploidies (<xref ref-type="supplementary-material" rid="supp3">Supplementary file 3</xref>). Twenty-nine mutants had a single mutation associated with fluorescence, 13 mutants had two associated mutations, and 4 mutants had three associated mutations. However, 8 of the 13 mutants with two associated mutations and all four mutants with three associated mutations showed linkage (genetic distance below 25 cM) between at least two of the mutations associated with fluorescence (<xref ref-type="supplementary-material" rid="supp4">Supplementary file 4</xref>, <xref ref-type="fig" rid="fig1s1">Figure 1—figure supplement 1</xref>), suggesting that only one of the linked mutations might impact fluorescence in each of these mutants. To determine whether one linked mutation was more likely to impact fluorescence than the others, we compared the magnitude of allele-frequency difference between the high and low fluorescence pools (estimated by the <italic>G</italic>-value) for each mutation. For 9 of the 12 mutants with linked mutations, we found that the mutation with the highest <italic>G</italic>-value was significantly more strongly associated with fluorescence than the linked mutation(s) (resampling test: p &lt; 0.05, <xref ref-type="supplementary-material" rid="supp4">Supplementary file 4</xref>), suggesting that this mutation was responsible for the fluorescence change. For the other three mutants, none of the linked mutations showed stronger evidence of impacting fluorescence than the others (resampling test: p &gt; 0.05, <xref ref-type="supplementary-material" rid="supp4">Supplementary file 4</xref>).</p><p>The remaining 36 mutants did not have any mutations significantly associated with fluorescence (<xref ref-type="supplementary-material" rid="supp2">Supplementary file 2</xref>, <xref ref-type="fig" rid="fig1s1">Figure 1—figure supplement 1</xref>). These mutants tended to show smaller changes in fluorescence than mutants with one or more associated mutations (<xref ref-type="fig" rid="fig2s2">Figure 2—figure supplement 2</xref>), suggesting that our power to map mutations causing 1% changes in fluorescence might have been lower than anticipated. These 36 mutants might also harbor multiple mutations with small effects on expression, each of which was below our detection threshold. Consistent with this possibility, we observed a small but significant correlation (<italic>r<sup>2</sup></italic> = 0.127, p = 0.03) between the total number of mutations in these 36 EMS mutants and their expression level (<xref ref-type="fig" rid="fig2s3">Figure 2—figure supplement 3</xref>). It is also possible that we failed to find associated mutations in some of these mutants because their change in fluorescence was initially overestimated by the ‘winner’s curse’ (<xref ref-type="bibr" rid="bib82">Xiao and Boehnke, 2009</xref>). Accordingly, 71% of mutants selected for mapping after two independent fluorescence screens had at least one mutation significantly associated with fluorescence compared to only 30% of mutants selected after a single fluorescence screen. Some changes in fluorescence observed in these 36 mutants might also have been caused by non-genetic variation and/or undetected mutations.</p></sec><sec id="s2-2"><title>Additional <italic>trans</italic>-regulatory mutations identified by sequencing candidate genes</title><p>We noticed in the BSA-seq data that three mutations increasing fluorescence more than 5% relative to the un-mutagenized progenitor strain mapped to two genes (<italic>ADE4</italic> and <italic>ADE5</italic>) in the same biochemical pathway (de novo purine biosynthesis) (<xref ref-type="supplementary-material" rid="supp2">Supplementary file 2</xref>). We therefore used Sanger sequencing to test whether these genes or other genes in this pathway were also mutated in 15 additional EMS mutants with fluorescence at least 5% higher than the progenitor strain. We first looked for mutations in <italic>ADE4</italic>, then <italic>ADE5</italic> if no mutation was found in <italic>ADE4</italic>, and then <italic>ADE6</italic> if no mutation was found in the other genes. At least one nonsynonymous mutation was identified by Sanger sequencing in one of these three genes in 14 of the 15 EMS mutants (green points in <xref ref-type="fig" rid="fig1">Figure 1C,E</xref>; <xref ref-type="supplementary-material" rid="supp5">Supplementary file 5</xref>, <xref ref-type="fig" rid="fig1s1">Figure 1—figure supplement 1</xref>). For the remaining mutant (brown point in <xref ref-type="fig" rid="fig1">Figure 1E</xref>), we sequenced a fourth purine biosynthesis gene, <italic>ADE8</italic>, but again found no mutation. In two additional EMS mutants with smaller increases in fluorescence (2.1% and 4.6%, purple points in <xref ref-type="fig" rid="fig1">Figure 1D,E</xref>) and a reddish color characteristic of <italic>ADE2</italic> loss of function mutants (<xref ref-type="bibr" rid="bib66">Roman, 1956</xref>), we found nonsynonymous mutations in <italic>ADE2</italic> by Sanger sequencing (<xref ref-type="supplementary-material" rid="supp5">Supplementary file 5</xref>, <xref ref-type="fig" rid="fig1s1">Figure 1—figure supplement 1</xref>). Follow-up experiments showed that mutations in <italic>ADE2</italic>, <italic>ADE5</italic>, and <italic>ADE6</italic> did not increase YFP fluorescence driven by two other promoters (<italic>P<sub>RNR1</sub></italic> and <italic>P<sub>STM1</sub></italic>), suggesting that mutations in the purine biosynthesis pathway affected expression of <italic>P<sub>TDH3</sub>-YFP</italic> through mechanisms mediated by the <italic>TDH3</italic> promoter rather than YFP (<xref ref-type="fig" rid="fig2s4">Figure 2—figure supplement 4</xref>). Taken together, these data suggest that genes in the purine biosynthesis pathway are the predominant mutational source of large increases in <italic>TDH3</italic> expression.</p></sec><sec id="s2-3"><title>Functional testing confirms effects of <italic>trans</italic>-regulatory mutations identified by genetic mapping and candidate gene sequencing</title><p>To determine whether mutations statistically associated with fluorescence in the BSA-seq data actually affected expression of <italic>P<sub>TDH3</sub>-YFP</italic>, we introduced 34 of the 67 associated mutations individually into the fluorescent progenitor strain using scarless genetic engineering approaches (<xref ref-type="supplementary-material" rid="supp6">Supplementary file 6</xref>, <xref ref-type="fig" rid="fig1s1">Figure 1—figure supplement 1</xref>). We also used scarless genome editing to create single-site mutants for 11 of the 17 additional mutations identified in purine biosynthesis genes by Sanger sequencing (<xref ref-type="supplementary-material" rid="supp5">Supplementary file 5</xref>, <xref ref-type="supplementary-material" rid="supp6">Supplementary file 6</xref>, <xref ref-type="fig" rid="fig1s1">Figure 1—figure supplement 1</xref>). Fluorescence of these engineered strains (called ‘single-site mutants’ hereafter) was then quantified by flow cytometry in parallel with fluorescence of the EMS mutant carrying the same associated mutation as well as the un-mutagenized progenitor strain, with four replicate populations analyzed for each genotype. Fluorescence values were then transformed into estimates of YFP abundance as described in the Methods.</p><p>Of the 24 mutations without linked variants in EMS mutants that were tested in single-site mutants, 23 (96%) caused a significant change in expression (p &lt; 0.05, permutation test, <xref ref-type="supplementary-material" rid="supp6">Supplementary file 6</xref>), suggesting a ~4% false positive rate in our BSA-Seq experiment. In addition, all 11 single-site mutants with mutations in purine biosynthesis genes identified by Sanger sequencing showed statistically significant effects on fluorescence relative to the un-mutagenized progenitor strain (all increased fluorescence, p &lt; 0.05, permutation test, <xref ref-type="supplementary-material" rid="supp6">Supplementary file 6</xref>). The remaining 10 mutations tested in single-site mutants were from five of the EMS mutants with two linked mutations associated with fluorescence. Each of these mutations was introduced separately into a single-site mutant to independently measure its effect on expression. For four of these five pairs of linked mutations, only one of the two single-site mutants showed a significant change in expression relative to the progenitor strain (<xref ref-type="fig" rid="fig2">Figure 2E</xref>). In each case, the single-site mutant and the EMS mutant showed changes in expression in the same direction relative to the progenitor strain (<xref ref-type="fig" rid="fig2">Figure 2E</xref>). The mutation affecting expression was always the mutation with the larger <italic>G</italic>-value in the BSA-Seq data, consistent with the results of the statistical tests described above (<xref ref-type="supplementary-material" rid="supp4">Supplementary file 4</xref>). In the last case (YPW54 in <xref ref-type="fig" rid="fig2">Figure 2E</xref>), both mutations affected expression in the single-site mutants, consistent with our inability to statistically predict which mutation was more likely to impact expression from the BSA-Seq data for this mutant as well as both mutations being nonsynonymous changes in the same gene (<italic>CHD1</italic>) (<xref ref-type="supplementary-material" rid="supp4">Supplementary file 4</xref>). The BSA-seq data also accurately predicted whether a mutation increased or decreased fluorescence for 27 (93%) of the 29 mutations with significant effects on fluorescence in single-site mutants (<xref ref-type="fig" rid="fig2">Figure 2F</xref>). For the other two mutations, effects on expression in the same direction were observed in the single-site mutants and the corresponding EMS mutants (<xref ref-type="supplementary-material" rid="supp5">Supplementary file 5</xref>), suggesting that the different growth conditions used for the mapping experiment (see Methods) might have modified the effects of these mutations.</p><p>Comparing <italic>P<sub>TDH3</sub>-YFP</italic> expression in the 40 single-site mutants that significantly altered fluorescence to that in the 40 EMS mutants from which these mutations were identified showed that expression was very similar overall between single-site and EMS mutants sharing the same mutation (<xref ref-type="fig" rid="fig2">Figure 2G</xref>, linear regression: <italic>r<sup>2</sup></italic> = 0.944, p = 2.4 x 10<sup>−25</sup>), although significant differences in expression were observed for some pairs (<xref ref-type="fig" rid="fig2">Figure 2G</xref>, <xref ref-type="fig" rid="fig2s5">Figure 2—figure supplement 5</xref>). The linear correlation between the expression of single-site mutants and EMS mutants remained strong when mutations identified by sequencing candidate genes (triangles in <xref ref-type="fig" rid="fig2">Figure 2G</xref>) were excluded (<italic>r<sup>2</sup></italic> = 0.854, p = 5.5 x 10<sup>−13</sup>). These data suggest that (1) the vast majority of the mutations we identified by genetic mapping and candidate gene sequencing do indeed have <italic>trans</italic>-regulatory effects on expression of <italic>P<sub>TDH3</sub>-YFP</italic> and (2) the majority of EMS mutants analyzed had a single mutation that was primarily, if not solely, responsible for the observed change in <italic>P<sub>TDH3</sub>-YFP</italic> expression.</p></sec><sec id="s2-4"><title>Properties of <italic>trans</italic>-regulatory mutations affecting expression driven by the <italic>TDH3</italic> promoter</title><p>In all, 69 mutations showed evidence of affecting <italic>P<sub>TDH3</sub>-YFP</italic> expression in <italic>trans</italic> (<xref ref-type="fig" rid="fig1s1">Figure 1—figure supplement 1</xref>), including 3 aneuploidies and 66 point mutations. Fifty-two of these mutations were identified by genetic mapping (<xref ref-type="supplementary-material" rid="supp7">Supplementary file 7</xref>) and 17 were identified by sequencing candidate genes (<xref ref-type="supplementary-material" rid="supp5">Supplementary file 5</xref>). Twelve of the mutations identified by genetic mapping were genetically linked to one or more other mutations but showed stronger evidence of affecting <italic>P<sub>TDH3</sub>-YFP</italic> expression than the linked mutation(s) in statistical and/or functional tests described above (<xref ref-type="supplementary-material" rid="supp3">Supplementary file 3</xref>). To identify trends in the properties of these 69 <italic>trans-</italic>regulatory mutations, we compared them to 1766 mutations considered non-regulatory regarding <italic>P<sub>TDH3</sub>-YFP</italic> expression because they showed no significant association with expression of the reporter gene in the BSA-Seq experiment (<italic>G</italic>-test: p &gt; 0.01, <xref ref-type="fig" rid="fig1s1">Figure 1—figure supplement 1</xref>). To be conservative, eight mutations that showed a marginally significant association with expression (<italic>G</italic>-test: 0.001 &lt; p &lt; 0.01) as well as 15 mutations associated with expression only because of genetic linkage were excluded from further analyses.</p><p>First, we asked whether the mutational spectra of <italic>trans-</italic>regulatory mutations differed from non-regulatory mutations (<xref ref-type="fig" rid="fig3">Figure 3A</xref>). We found that G:C to A:T transitions most commonly introduced by EMS occurred at similar frequencies in the two groups (<italic>G</italic>-test, p = 0.84). No indels were associated with expression in the BSA-seq data (<xref ref-type="supplementary-material" rid="supp7">Supplementary file 7</xref>), which was not statistically different from the frequency of indels among non-regulatory mutations (0% vs 2.7%, <italic>G</italic>-test, p = 0.056). By contrast, aneuploidies were highly over-represented in the set of <italic>trans-</italic>regulatory mutations since all three extra copies of a chromosome observed in the BSA-Seq data were found to be associated with fluorescence (<italic>G</italic>-test, p = 8.6 x 10<sup>−6</sup>); a similar overrepresentation was observed when considering only mutations identified by BSA-Seq (<xref ref-type="fig" rid="fig3s1">Figure 3—figure supplement 1A</xref>; <italic>G</italic>-test, p = 3.5 x 10<sup>−6</sup>). We also found a significant difference in the genomic distribution of the two sets of mutations (<italic>G</italic>-test, p = 2.4 x 10<sup>−3</sup>), with non-regulatory mutations appearing to be randomly distributed throughout the genome but <italic>trans-</italic>regulatory mutations enriched on chromosomes VII and XIII (<xref ref-type="fig" rid="fig3">Figure 3B</xref>, <xref ref-type="fig" rid="fig3s2">Figure 3—figure supplement 2</xref>). However, these two chromosomes contain the purine biosynthesis genes in which multiple <italic>trans</italic>-regulatory mutations were identified, and there was no significant difference in genomic distributions between <italic>trans</italic>-regulatory and non-regulatory mutations when mutations in purine biosynthesis genes were excluded (<italic>G</italic>-test, p = 0.35) or when mutations identified by direct sequencing of candidate genes were excluded (<xref ref-type="fig" rid="fig3s1">Figure 3—figure supplement 1B</xref>; <italic>G</italic>-test, p = 0.22).</p><fig-group><fig id="fig3" position="float"><label>Figure 3.</label><caption><title>Contrasting properties of <italic>trans</italic>-regulatory and non-regulatory mutations.</title><p>(<bold>A</bold>) Proportions of different types of mutations in a set of 1766 non-regulatory mutations (blue) and in a set of 69 <italic>trans</italic>-regulatory mutations (orange). Numbers of mutations are indicated above bars. (<bold>B</bold>) Distributions of non-regulatory and <italic>trans</italic>-regulatory point mutations along the yeast genome. A total of 1766 non-regulatory mutations are shown in blue, 44 <italic>trans</italic>-regulatory mutations that were identified from the collections of unenriched mutants in <xref ref-type="bibr" rid="bib53">Metzger et al., 2016</xref> are shown in red and 22 <italic>trans</italic>-regulatory mutations that were identified from the collections of mutants enriched for large expression changes in <xref ref-type="bibr" rid="bib20">Gruber et al., 2012</xref> and in <xref ref-type="bibr" rid="bib53">Metzger et al., 2016</xref> are shown in green. (<bold>C</bold>) Proportions of non-regulatory (left) and <italic>trans</italic>-regulatory (right) mutations affecting either coding sequences, introns or intergenic regions. (<bold>D</bold>) Proportions of coding non-regulatory (left) and coding <italic>trans</italic>-regulatory (right) mutations that either introduce an early stop codon (nonsense), that substitute one amino acid for another (nonsynonymous) or that do not change the amino acid sequence (synonymous). (<bold>E</bold>) Frequency of all amino acid changes induced by <italic>trans</italic>-regulatory mutations as compared to non-regulatory mutations. Each entry of the table represents the difference of frequency (percentage) between non-regulatory and <italic>trans</italic>-regulatory mutations that are changing the amino acid shown on the y-axis into the amino acid shown on the x-axis. For instance, the −6 on the first row indicates that the proportion of mutations changing an Alanine into a Threonine is 6% lower among <italic>trans</italic>-regulatory mutations than among non-regulatory mutations. Shades of red: amino acid changes underrepresented in the set of <italic>trans</italic>-regulatory mutations. Shades of green: amino acid changes overrepresented in the set of <italic>trans</italic>-regulatory mutations. White: amino acid changes equally represented in the <italic>trans</italic>-regulatory and non-regulatory sets of mutations. Gray: amino acid changes not observed in the sets of <italic>trans</italic>-regulatory and non-regulatory mutations. (<bold>B–E</bold>) The three aneuploidies were excluded for these plots. (<bold>D,E</bold>) Non-coding mutations were excluded for these plots.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-67806-fig3-v1.tif"/></fig><fig id="fig3s1" position="float" specific-use="child-fig"><label>Figure 3—figure supplement 1.</label><caption><title>Contrasting properties of non-regulatory and <italic>trans</italic>-regulatory mutations identified by BSA-Seq and of <italic>trans</italic>-regulatory mutations identified by Sanger sequencing of candidate genes.</title><p>(<bold>A</bold>) Proportions of different types of mutations observed among 1766 non-regulatory mutations (blue), among 52 <italic>trans</italic>-regulatory mutations identified by BSA-Seq (red) and among 17 trans-regulatory mutations identified by Sanger sequencing of candidate genes. Numbers of mutations are indicated above bars. (<bold>B</bold>) Distributions of non-regulatory and <italic>trans</italic>-regulatory point mutations along the yeast genome. A total of 1766 non-regulatory mutations are shown in blue, 49 <italic>trans</italic>-regulatory mutations that were identified by BSA-Seq are shown in red and 17 <italic>trans</italic>-regulatory mutations that were identified by Sanger sequencing are shown in green. (<bold>C</bold>) Proportions of non-regulatory mutations (left), <italic>trans</italic>-regulatory mutations identified by BSA-Seq (upper right) and <italic>trans</italic>-regulatory mutations identified by Sanger sequencing (bottom right) that affect either coding sequences, introns or intergenic regions. (<bold>D</bold>) Proportions of coding non-regulatory mutations (left), coding <italic>trans</italic>-regulatory mutations identified by BSA-Seq (upper right) and coding <italic>trans-</italic>regulatory mutations identified by Sanger sequencing (bottom right) that either introduce an early stop codon (nonsense), that substitute one amino acid for another (nonsynonymous) or that do not change the amino acid sequence (synonymous). (<bold>E</bold>) Frequency of all amino acid changes induced by <italic>trans</italic>-regulatory mutations identified by BSA-Seq as compared to non-regulatory mutations. Each entry of the table represents the difference of frequency (percentage) between non-regulatory and <italic>trans</italic>-regulatory mutations that are changing the amino acid shown on the y-axis into the amino acid shown on the x-axis. Shades of red: amino acid changes underrepresented in the set of <italic>trans</italic>-regulatory mutations identified by BSA-Seq. Shades of green: amino acid changes overrepresented in the set of <italic>trans</italic>-regulatory mutations identified by BSA-Seq. White: amino acid changes equally represented in the <italic>trans</italic>-regulatory and non-regulatory sets of mutations. Gray: amino acid changes not observed in the sets of <italic>trans</italic>-regulatory and non-regulatory mutations. (<bold>B–E</bold>) The three aneuploidies were excluded for these plots. (<bold>D,E</bold>) Non-coding mutations were excluded for these plots.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-67806-fig3-figsupp1-v1.tif"/></fig><fig id="fig3s2" position="float" specific-use="child-fig"><label>Figure 3—figure supplement 2.</label><caption><title>Distributions of <italic>trans</italic>-regulatory and non-regulatory mutations among chromosomes.</title><p>1766 non-regulatory mutations are shown in blue and 69 <italic>trans</italic>-regulatory mutations are shown in orange, among which 52 mutations were identified by BSA-Seq (shown in red) and 17 mutations were identified by Sanger sequencing of candidate genes (shown in green). <italic>Trans</italic>-regulatory mutations were significantly enriched on chromosome VII that contained the purine biosynthesis genes <italic>ADE5</italic> and <italic>ADE6</italic> in which several mutations were identified (24.3% of <italic>trans</italic>-regulatory mutations located on chromosome VII <italic>vs</italic> 9.3% of non-regulatory mutations; <italic>G</italic>-test, p = 3.4 x 10<sup>−4</sup>). <italic>Trans</italic>-regulatory mutations were also enriched on chromosome XIII that contained the purine synthesis gene <italic>ADE4</italic>, although this enrichment was not statistically significant (13.0% of <italic>trans</italic>-regulatory mutations located on chromosome XIII <italic>vs</italic> 7.8% of non-regulatory mutations; <italic>G</italic>-test, p = 0.15).</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-67806-fig3-figsupp2-v1.tif"/></fig><fig id="fig3s3" position="float" specific-use="child-fig"><label>Figure 3—figure supplement 3.</label><caption><title>Statistical significance of the enrichment and depletion of amino acid changes induced by <italic>trans</italic>-regulatory mutations.</title><p>Permutations tests were used to assess the statistical significance of the frequency differences between non-regulatory and <italic>trans</italic>-regulatory mutations shown on <xref ref-type="fig" rid="fig3">Figure 3E</xref>. Each number represents the negative logarithm (base-10) of the p-value obtained using a permutation test to compare the frequency of changing the amino acid on the y-axis to the amino acid shown on the x-axis between non-regulatory and <italic>trans</italic>-regulatory mutations. Green color intensity scales with the negative logarithm of p-values. White: amino acid changes equally represented in the <italic>trans</italic>-regulatory and non-regulatory sets of mutations. Gray: amino acid changes not observed in the sets of <italic>trans</italic>-regulatory and non-regulatory mutations.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-67806-fig3-figsupp3-v1.tif"/></fig><fig id="fig3s4" position="float" specific-use="child-fig"><label>Figure 3—figure supplement 4.</label><caption><title>Statistical significance of the enrichment and depletion of amino acid changes induced by <italic>trans</italic>-regulatory mutations identified by BSA-Seq.</title><p>Permutations tests were used to assess the statistical significance of the frequency differences between non-regulatory and <italic>trans</italic>-regulatory mutations shown on <xref ref-type="fig" rid="fig3s1">Figure 3—figure supplement 1E</xref>. Each number represents the negative logarithm (base-10) of the p-value obtained using a permutation test to compare the frequency of changing the amino acid on the y-axis to the amino acid shown on the x-axis between non-regulatory and <italic>trans</italic>-regulatory mutations. Green color intensity scales with the negative logarithm of p-values. White: amino acid changes equally represented in the <italic>trans</italic>-regulatory and non-regulatory sets of mutations. Gray: amino acid changes not observed in the sets of <italic>trans</italic>-regulatory and non-regulatory mutations.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-67806-fig3-figsupp4-v1.tif"/></fig></fig-group><p><italic>Trans</italic>-regulatory mutations are often assumed to be located in coding sequences, but they can also be located in non-coding, presumably <italic>cis</italic>-regulatory, sequences of <italic>trans</italic>-acting genes (<xref ref-type="bibr" rid="bib22">Hill et al., 2021</xref>). We therefore asked whether <italic>trans-</italic>regulatory mutations affecting <italic>P<sub>TDH3</sub>-YFP</italic> expression were more often found in coding or non-coding regions of the genome than expected by chance. Of the 1766 non-regulatory mutations, 1257 (71.3%) were coding mutations located in exons, and 506 (28.7%) were non-coding mutations located in intergenic (n = 500) or intronic (n = 6) regions (<xref ref-type="fig" rid="fig3">Figure 3C</xref>). This paucity of mutations in introns is consistent with the rarity of introns in <italic>S. cerevisiae</italic>, and the overall frequency of non-coding mutations (28.7%) is similar to the fraction of the <italic>S. cerevisiae</italic> genome (30.6% of 12.1 Mb) considered non-coding (<ext-link ext-link-type="uri" xlink:href="https://www.yeastgenome.org/">https://www.yeastgenome.org/</ext-link>). By contrast, of the 66 <italic>trans</italic>-regulatory point mutations, only one was located in a non-coding sequence (<xref ref-type="fig" rid="fig3">Figure 3C</xref>). This non-coding mutation was located in the intergenic sequence between <italic>IOC2</italic> and <italic>KIN2</italic>, presumably affecting expression of one or both genes with a downstream effect on <italic>P<sub>TDH3</sub>-YFP</italic> expression. The three aneuploidies were excluded from this and subsequent analyses because they affected both coding and non-coding sequences of a large number of genes. The underrepresentation of non-coding changes among regulatory mutations was statistically significant (1.5% of <italic>trans-</italic>regulatory mutations are non-coding <italic>vs</italic> 28.4% of non-regulatory mutations; <italic>G</italic>-test, p = 4.3 x 10<sup>−9</sup>), even when excluding mutations identified by sequencing candidate genes (<xref ref-type="fig" rid="fig3s1">Figure 3—figure supplement 1C</xref>; <italic>G</italic>-test, p = 9.1 x 10<sup>−7</sup>). These observations suggest that new <italic>trans</italic>-regulatory mutations affecting <italic>P<sub>TDH3</sub>-YFP</italic> expression by more than 3% (i.e. fluorescence changes greater than 1%) are more likely to alter coding than non-coding sequences. This enrichment in coding sequences might be because coding sequences tend to have a higher density of functional sites than non-coding sequences.</p><p>Finally, we examined how <italic>trans</italic>-regulatory mutations located in coding sequences impacted the amino acid sequences of the corresponding proteins. Among mutations identified in coding sequences, 100% of the 65 <italic>trans-</italic>regulatory mutations changed the amino acid sequence of proteins compared to only 70% of 1257 non-regulatory mutations (<xref ref-type="fig" rid="fig3">Figure 3D,G</xref> -test, p = 1.4 x 10<sup>−4</sup>). Limiting this analysis to the 48 <italic>trans-</italic>regulatory mutations identified by BSA-seq also showed an enrichment of mutations changing the amino acid sequence of proteins (<xref ref-type="fig" rid="fig3s1">Figure 3—figure supplement 1D,G</xref> -test, p = 5.6 x 10<sup>−6</sup>). This difference was primarily driven by mutations that introduced stop codons (nonsense mutations) rather than mutations that substituted one amino acid for another (nonsynonymous mutations): 20% of <italic>trans</italic>-regulatory mutations in coding sequences were nonsense mutations <italic>versus</italic> 3% of non-regulatory mutations (<xref ref-type="fig" rid="fig3">Figure 3D</xref>; <italic>G</italic>-test, p = 4.8 x 10<sup>−6</sup>), and 80% of <italic>trans</italic>-regulatory mutations were nonsynonymous versus 67% of non-regulatory mutations (<xref ref-type="fig" rid="fig3">Figure 3D</xref>; <italic>G</italic>-test, p = 0.07). A similar pattern was observed when considering only <italic>trans</italic>-regulatory mutations identified by BSA-Seq (<xref ref-type="fig" rid="fig3s1">Figure 3—figure supplement 1D</xref>). Nonsense mutations always altered an arginine, glutamine, or tryptophan codon (<xref ref-type="fig" rid="fig3">Figure 3E</xref>), consistent with the structure of the genetic code and the types of mutations induced by EMS (Figure 3—figure supplement 2 in <xref ref-type="bibr" rid="bib53">Metzger et al., 2016</xref>). For nonsynonymous mutations, two types of amino acid changes were particularly enriched among <italic>trans</italic>-regulatory mutations (<xref ref-type="fig" rid="fig3">Figure 3E</xref>; <xref ref-type="fig" rid="fig3s3">Figure 3—figure supplement 3</xref>): 26.2% of <italic>trans</italic>-regulatory mutations changed glycine to aspartic acid <italic>versus</italic> 5.2% of non-regulatory mutations (permutation test, p &lt; 10<sup>−4</sup>), and 10.8% of <italic>trans</italic>-regulatory mutations changed glycine to glutamic acid versus 2.7% of non-regulatory mutations (permutation test, p = 0.0042). As a consequence, mutations altering glycine codons were strongly over-represented in general among <italic>trans</italic>-regulatory mutations (49.2% of <italic>trans</italic>-regulatory mutations <italic>vs</italic> 14.5% of non-regulatory mutations in coding sequences; permutation test, p &lt; 10<sup>−4</sup>). This over-representation remained significant after excluding mutations identified by Sanger sequencing (<xref ref-type="fig" rid="fig3s1">Figure 3—figure supplement 1E</xref>, <xref ref-type="fig" rid="fig3s4">Figure 3—figure supplement 4</xref>; 41.7% of <italic>trans</italic>-regulatory mutations altering glycine vs 14.5% of non-regulatory mutations, p = 10<sup>−4</sup>). This pattern may be observed because glycine is the smallest amino acid, making its substitution likely to modify protein structure (<xref ref-type="bibr" rid="bib4">Bhate et al., 2002</xref>; <xref ref-type="bibr" rid="bib56">Miller, 2007</xref>). Indeed, glycine is one of the three amino acids with the lowest experimental exchangeability (<xref ref-type="bibr" rid="bib84">Yampolsky and Stoltzfus, 2005</xref>) and mutations affecting glycine codons are enriched among mutations causing human diseases (<xref ref-type="bibr" rid="bib33">Khan and Vihinen, 2007</xref>; <xref ref-type="bibr" rid="bib57">Molnár et al., 2016</xref>; <xref ref-type="bibr" rid="bib81">Vitkup et al., 2003</xref>).</p></sec><sec id="s2-5"><title>Regulatory mutations are enriched in a predicted <italic>TDH3</italic> regulatory network</title><p>Because of the key role transcription factors play in the regulation of gene expression, and because transcription factors have been shown to be a source of <italic>trans</italic>-regulatory variation in natural populations (<xref ref-type="bibr" rid="bib1">Albert et al., 2018</xref>; <xref ref-type="bibr" rid="bib39">Lewis et al., 2014</xref>), we asked whether <italic>trans</italic>-regulatory mutations affecting <italic>P<sub>TDH3</sub>-YFP</italic> expression were enriched in genes encoding transcription factors. We found that 5 (7.7%) of the 65 <italic>trans</italic>-regulatory coding mutations mapped to the coding sequence of one of the 212 genes predicted to encode a transcription factor in the YEASTRACT database (<xref ref-type="bibr" rid="bib76">Teixeira et al., 2018</xref>), but this was not significantly more than the 5.6% of non-regulatory coding mutations mapping to these genes (<italic>G</italic>-test: p = 0.52). <italic>Trans</italic>-regulatory coding mutations were also not significantly enriched in transcription factor genes when we excluded the 17 mutations identified by Sanger sequencing (<italic>G</italic>-test: p = 0.22). Not all transcription factors are expected to regulate expression of <italic>TDH3</italic>, however, so we also tested for enrichment of <italic>trans</italic>-regulatory mutations among transcription factors specifically predicted to regulate <italic>TDH3</italic>.</p><p>Using information consolidated in the YEASTRACT database (<xref ref-type="bibr" rid="bib76">Teixeira et al., 2018</xref>) that supports evidence of a transcription factor binding to a gene’s promoter and regulating its expression, we constructed a network (<xref ref-type="fig" rid="fig4">Figure 4</xref>) of potential direct regulators of <italic>TDH3</italic> as well as potential direct regulators of these direct regulators (1st and 2nd level regulators of <italic>TDH3</italic>) and asked how often the <italic>trans-</italic>regulatory mutations we identified mapped to these genes. We found that four <italic>trans</italic>-regulatory mutations mapped to three genes in this network, with two mutations affecting the 1st level regulator <italic>TYE7</italic>, one mutation affecting the 1st level regulator <italic>GCR2</italic>, and one mutation affecting the 2nd level regulator <italic>TUP1</italic> (<xref ref-type="supplementary-material" rid="supp7">Supplementary file 7</xref>). This number of mutations mapping to genes in the predicted <italic>TDH3</italic> regulatory network was 12-fold greater than expected by chance (6.1% for <italic>trans</italic>-regulatory <italic>vs</italic> 0.5% for non-regulatory mutations; <italic>G</italic>-test, p = 0.0037), or 16-fold greater than expected by chance when excluding mutations identified by Sanger sequencing (8.2% for <italic>trans</italic>-regulatory <italic>vs</italic> 0.5% for non-regulatory mutations; <italic>G</italic>-test, p = 0.0024). Therefore, the inferred regulatory network had predictive power as expected, but the vast majority of <italic>trans-</italic>regulatory coding mutations (61 of 65, or 94%) mapped to genes outside of this network. Only one of these other <italic>trans-</italic>regulatory mutations mapped to a transcription factor. This mutation was a nonsynonymous substitution affecting <italic>ROX1</italic>, which is predicted in the YEASTRACT database to directly regulate expression of the indirect <italic>TDH3</italic> regulator <italic>TUP1.</italic> In other words, <italic>ROX1</italic> is predicted by existing functional genomic data to be a 3rd level regulator of <italic>TDH3</italic> (<xref ref-type="fig" rid="fig4">Figure 4</xref>). With no other transcription factors harboring a <italic>trans-</italic>regulatory mutation in our dataset, this result suggests that mutations in transcription factors located more than three levels away from <italic>TDH3</italic> in its transcriptional regulatory network are unlikely to be sources of new expression changes driven by the <italic>TDH3</italic> promoter.</p><fig id="fig4" position="float"><label>Figure 4.</label><caption><title>Mutations mapping to a predicted <italic>TDH3</italic> regulatory network.</title><p>The network of inferred interactions between <italic>TDH3</italic> and transcription factors regulating its expression was established using the YEASTRACT repository (<xref ref-type="bibr" rid="bib76">Teixeira et al., 2018</xref>). First level regulators (dark gray boxes) are transcription factors with evidence of binding to the <italic>TDH3</italic> promoter and regulating its expression. Second level regulators (light gray boxes) are transcription factors with evidence of binding to the promoter of at least one first level regulator and regulating its expression. Green arrows: evidence for activation of expression. Red arrows: evidence for inhibition of expression. Black arrows: unknown direction of regulation. Non-regulatory and <italic>trans</italic>-regulatory mutations identified in the network are represented by blue and orange stars, respectively, near the affected genes. <italic>ROX1</italic>, inferred to be a third level regulator, is also shown because a <italic>trans</italic>-regulatory mutation was identified in its coding sequence.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-67806-fig4-v1.tif"/></fig></sec><sec id="s2-6"><title>Deleterious effects of mutations in two direct regulators of <italic>TDH3</italic></title><p>Transcription factors encoded by the <italic>TYE7</italic> and <italic>GCR2</italic> genes found to harbor <italic>trans</italic>-regulatory mutations affecting expression of <italic>P<sub>TDH3</sub>-YFP</italic> are known to regulate the expression of glycolytic genes (including <italic>TDH3</italic>) by forming a complex with transcription factors encoded by the <italic>RAP1</italic> and <italic>GCR1</italic> genes (<xref ref-type="bibr" rid="bib70">Shively et al., 2019</xref>). Rap1p (<xref ref-type="bibr" rid="bib83">Yagi et al., 1994</xref>) and Gcr1p (<xref ref-type="bibr" rid="bib28">Huie et al., 1992</xref>) are both known to bind directly to the <italic>TDH3</italic> promoter (<xref ref-type="fig" rid="fig5">Figure 5A</xref>), and mutations in these binding sites cause large decreases in <italic>TDH3</italic> expression (<xref ref-type="bibr" rid="bib52">Metzger et al., 2015</xref>). These observations strongly suggest that mutations in <italic>RAP1</italic> and <italic>GCR1</italic> should also cause detectable changes in <italic>TDH3</italic> expression, yet no mutations were observed in these genes in our set of <italic>trans</italic>-regulatory mutations. To investigate why we did not recover <italic>trans-</italic>regulatory mutations in <italic>RAP1</italic> or <italic>GCR1</italic>, we used error-prone PCR to generate mutant alleles of these genes with mutations in either the promoter or coding sequence of <italic>RAP1</italic> or the second exon of <italic>GCR1</italic>, which includes 99.7% of the <italic>GCR1</italic> coding sequence (<xref ref-type="fig" rid="fig5">Figure 5B</xref>). Hundreds of these <italic>RAP1</italic> and <italic>GCR1</italic> mutant alleles were then introduced individually into the un-mutagenized strain carrying the <italic>P<sub>TDH3</sub>-YFP</italic> reporter gene using CRISPR/Cas9-guided allelic replacement. Sequencing the mutated regions of <italic>RAP1</italic> and <italic>GCR1</italic> in a random subset of transformants showed that each strain harbored an average of 1.8 mutations in the <italic>RAP1</italic> gene (<xref ref-type="fig" rid="fig5">Figure 5C</xref>) or 2.4 mutations in the <italic>GCR1</italic> gene (<xref ref-type="fig" rid="fig5">Figure 5D</xref>). As expected for PCR-based mutagenesis, the number of mutations per strain appeared to follow a Poisson distribution both for <italic>RAP1</italic> mutants (<xref ref-type="fig" rid="fig5">Figure 5C</xref>, Chi-square goodness of fit, p = 0.14) and <italic>GCR1</italic> mutants (<xref ref-type="fig" rid="fig5">Figure 5D</xref>, Chi-square goodness of fit, p = 0.79).</p><fig id="fig5" position="float"><label>Figure 5.</label><caption><title>Impact of mutations in two direct regulators of the <italic>TDH3</italic> promoter.</title><p>(<bold>A</bold>) Schematics of the <italic>P<sub>TDH3</sub>-YFP</italic> reporter gene with locations of three known binding sites for transcription factors Rap1p (purple) and Gcr1p (green) shown in the <italic>TDH3</italic> promoter. (<bold>B</bold>) Regions of <italic>RAP1</italic> (purple) and <italic>GCR1</italic> (green) genes that were subjected to random mutagenesis using error-prone PCR. 470 <italic>RAP1</italic> mutants and 220 <italic>GCR1</italic> mutants were obtained by integration of random PCR fragments at the native <italic>RAP1</italic> or <italic>GCR1</italic> loci using CRISPR/Cas9 allelic replacement. (<bold>C–D</bold>) Distributions of the number of mutations per strain identified by Sanger sequencing the mutated regions of (<bold>C</bold>) <italic>RAP1</italic> in 27 strains or (<bold>D</bold>) <italic>GCR1</italic> in 18 strains. These data are shown in histograms. Blue curves: Poisson distribution with the same mean as observed in data. Red dotted line: Mean number of mutations among sequenced strains. (<bold>E–F</bold>) Distributions of <italic>P<sub>TDH3</sub>-YFP</italic> expression changes relative to the un-mutagenized reporter strain measured in four replicate samples for (<bold>E</bold>) the 470 <italic>RAP1</italic> mutants or (<bold>F</bold>) the 220 <italic>GCR1</italic> mutants. Fluorescence measures were transformed to be linearly related with <italic>YFP</italic> mRNA levels (see Methods). Red bars: Mutants with significant decrease in median expression greater than 3% relative to the un-mutagenized strain (permutation test, p &lt; 0.05). Blue bars: Mutants with significant increase in median expression greater than 3% relative to the un-mutagenized strain (permutation test, p &lt; 0.05). Pie charts: Proportions of mutants with significant increase in expression (blue), significant decrease in expression (red) and no significant change in expression (gray) relative to the un-mutagenized strain. (<bold>G</bold>) Relationship between changes in <italic>P<sub>TDH3</sub>-YFP</italic> expression levels (x-axis) and fitness (y-axis) measured in 62 <italic>GCR1</italic> mutants. Expression changes and fitness are both expressed relative to the un-mutagenized strain. Gray dotted lines: Expression change and fitness of the un-mutagenized strain. Error bars: 95% confidence intervals of expression changes and fitness measures obtained from four replicate populations of each mutant. The black dotted line represents a LOESS regression of fitness on median expression with a smoothing parameter of 1% and 95% confidence intervals of the estimates shown as a gray shaded area.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-67806-fig5-v1.tif"/></fig><p>Among the <italic>RAP1</italic> mutant strains, only 9.1% (43 of 470 strains) showed a significant change in <italic>P<sub>TDH3</sub>-YFP</italic> expression greater than 3% (corresponding to a ~1% change in fluorescence) relative to the un-mutagenized progenitor strain (<xref ref-type="fig" rid="fig5">Figure 5E</xref>), suggesting that most EMS mutants harboring coding mutations in <italic>RAP1</italic> would have been excluded from our mapping study. In addition, the strongest decrease in <italic>P<sub>TDH3</sub>-YFP</italic> expression observed among <italic>RAP1</italic> mutants (17%) was substantially smaller than the strongest decrease in expression caused by mutating the RAP1-binding site in the <italic>TDH3</italic> promoter (57.5% reported in <xref ref-type="bibr" rid="bib11">Duveau et al., 2018</xref>), suggesting that even this most severe phenotype was not caused by a null allele of <italic>RAP1</italic>. To test this hypothesis, we used site-directed mutagenesis to alter five amino acids (one at a time) in Rap1p expected to disrupt DNA binding based on the crystal structure of Rap1p complexed with DNA (<xref ref-type="bibr" rid="bib34">Konig et al., 1996</xref>). In each case, we obtained by PCR a DNA fragment containing either a synonymous mutation in the codon corresponding to the amino acid (which should not affect the DNA binding of Rap1p) or one of two nonsynonymous mutations, with one nonsynonymous mutation more likely to alter protein function than the other (<xref ref-type="bibr" rid="bib84">Yampolsky and Stoltzfus, 2005</xref>). We then used CRISPR/Cas9 allele replacement to introduce each mutation into the yeast genome and sequenced 10 independent clones from each transformation to determine if the mutation was introduced in the <italic>RAP1</italic> coding sequence as intended. All five synonymous mutations were observed in several of the clones sequenced, but 7 of the 10 nonsynonymous mutations were never recovered (<xref ref-type="supplementary-material" rid="supp8">Supplementary file 8</xref>). This outcome suggests that nonsynonymous mutations altering the DNA binding of Rap1p are lethal or nearly lethal, making them unlikely to have been recovered in a mutagenesis screen. Indeed, Rap1p is known to be an essential, pleiotropic transcription factor playing critical roles in regulating expression of glycolytic genes like <italic>TDH3</italic> as well as ribosomal proteins and genes required for mating (reviewed in <xref ref-type="bibr" rid="bib62">Piña et al., 2003</xref>). Taken together, these data indicate that <italic>RAP1</italic> mutations are unlikely to be common sources of variation in expression driven by the <italic>TDH3</italic> promoter.</p><p>For the <italic>GCR1</italic> mutant strains, 37.7% showed a significant change in <italic>P<sub>TDH3</sub>-YFP</italic> expression greater than 3% relative to the un-mutagenized progenitor strain (<xref ref-type="fig" rid="fig5">Figure 5F</xref>). Several of these mutant alleles decreased the expression driven by the <italic>TDH3</italic> promoter by ~80%, which is similar to the previously reported effects of mutations in the Gcr1p binding sites of the <italic>TDH3</italic> promoter (<xref ref-type="bibr" rid="bib52">Metzger et al., 2015</xref>), suggesting that they were null alleles. Indeed, resequencing these large effect alleles revealed that one of them had a single nucleotide insertion in the 28th codon of the <italic>GCR1</italic> ORF, which led to a frame shift eliminating 96% of amino acids (757 of 785) from Gcr1p. Because Gcr1p regulates expression of many glycolytic genes (<xref ref-type="bibr" rid="bib78">Uemura et al., 1997</xref>) and <italic>GCR1</italic> deletion has been reported to cause severe growth defects in fermentable carbon source environments (<xref ref-type="bibr" rid="bib6">Clifton et al., 1978</xref>; <xref ref-type="bibr" rid="bib25">Hossain et al., 2016</xref>; <xref ref-type="bibr" rid="bib43">López and Baker, 2000</xref>), we hypothesized that the fitness effects of mutations in <italic>GCR1</italic> might also have caused them to be underrepresented in the population from which the EMS mutants analyzed were derived. To test this hypothesis, we measured the relative fitness of 62 of the 220 <italic>GCR1</italic> mutants, including all mutants with decreased <italic>P<sub>TDH3</sub>-YFP</italic> expression. <italic>GCR1</italic> mutants causing the largest changes in <italic>P<sub>TDH3</sub>-YFP</italic> expression showed strong defects in growth rate; however, several <italic>GCR1</italic> mutants with changes in <italic>P<sub>TDH3</sub>-YFP</italic> expression greater than 3% did not strongly affect fitness (<xref ref-type="fig" rid="fig5">Figure 5G</xref>). This observation suggests that some of the coding mutations in <italic>GCR1</italic> decreasing <italic>P<sub>TDH3</sub>-YFP</italic> expression could have been sampled among the EMS mutants used for mapping. We therefore conclude that mutations in <italic>GCR1</italic> were most likely not recovered in our set of regulatory mutations because of the wide diversity of mutations that can affect <italic>TDH3</italic> expression and the limited number of EMS mutants included in the mapping experiment.</p></sec><sec id="s2-7"><title>Properties of genes harboring regulatory mutations</title><p>With only 5 of the 65 <italic>trans</italic>-regulatory point mutations in coding sequences mapping to transcription factors, we used gene ontology (GO) analysis to examine the types of genes harboring <italic>trans</italic>-regulatory mutations affecting <italic>P<sub>TDH3</sub>-YFP</italic> expression more systematically. In all, these 65 mutations mapped to 42 different genes, with nine genes affected by more than one mutation, 4 of which were genes involved in the de novo purine biosynthesis pathway (<xref ref-type="fig" rid="fig6">Figure 6A</xref>). Several gene ontology terms were significantly enriched among genes affected by <italic>trans</italic>-regulatory mutations relative to genes affected by non-regulatory mutations. <xref ref-type="supplementary-material" rid="supp9">Supplementary file 9</xref> includes all enriched GO terms, whereas <xref ref-type="fig" rid="fig6">Figure 6B</xref> only includes enriched GO terms that are not parent to other GO terms in the GO hierarchy. Excluding mutations identified by sequencing candidate genes had a negligible impact on the outcome of the GO term analysis, with more than 96% of overlap between the GO terms found to be enriched before and after excluding mutations identified by Sanger sequencing (<xref ref-type="supplementary-material" rid="supp8">Supplementary file 8</xref>). Of the 33 GO terms enriched for <italic>trans</italic>-regulatory mutations shown in <xref ref-type="fig" rid="fig6">Figures 6B,</xref> 11 terms (including 13 of the 42 genes with <italic>trans</italic>-regulatory mutations) were related to chromatin structure (<xref ref-type="fig" rid="fig6">Figure 6B</xref>), which is known to play an important role in the regulation of gene expression (<xref ref-type="bibr" rid="bib40">Li et al., 2007</xref>). An additional five GO terms (including six genes with <italic>trans</italic>-regulatory mutations) were related to metabolism, and four terms (including nine genes with <italic>trans</italic>-regulatory mutations) were related to transcriptional regulation (<xref ref-type="fig" rid="fig6">Figure 6B</xref>). Three GO terms related to glucose signaling, including regulation of transcription by glucose, carbohydrate transmembrane transport and glucose metabolic process, were also significantly enriched for genes affected by <italic>trans</italic>-regulatory mutations (<xref ref-type="fig" rid="fig6">Figure 6B</xref>). When we broadened this category of genes based on a review of glucose signaling (<xref ref-type="bibr" rid="bib67">Santangelo, 2006</xref>), the enrichment included five genes implicated in glucose signaling (<xref ref-type="supplementary-material" rid="supp10">Supplementary file 10</xref>; 12.2% of genes affected by <italic>trans</italic>-regulatory mutations were involved in glucose signaling <italic>vs</italic> 2.7% of genes affected by non-regulatory mutations; Fisher’s exact test: p = 6.2 x 10<sup>−3</sup>).</p><fig id="fig6" position="float"><label>Figure 6.</label><caption><title>Properties of genes with coding mutations altering <italic>P<sub>TDH3</sub>-YFP</italic> expression level.</title><p>(<bold>A</bold>) Proportion of genes with one or more mutations identified among EMS mutants. Mutations in intergenic regions were excluded from this analysis. Orange bars include genes harboring one or more of the 65 <italic>trans</italic>-regulatory mutations identified in coding sequences. Blue bars include genes harboring one or more of 65 non-regulatory mutations randomly chosen among the set of 1095 non-regulatory mutations observed in coding sequences. The number of genes hit by 1–8 mutations is indicated above the corresponding bar. For blue bars, this number represents the mean number of genes obtained from 1000 random sets of 65 non-regulatory mutations. The names of genes with at least two <italic>trans</italic>-regulatory mutations identified among mutants are indicated above the bars. <italic>FTR1</italic> and <italic>CCC2</italic> are involved in iron homeostasis, <italic>ADE2,4,5,6</italic> are involved in de novo purine biosynthesis, <italic>NAM7</italic> is involved in nonsense-mediated mRNA decay, <italic>CHD1</italic> is involved in chromatin regulation and <italic>TYE7</italic> encodes a transcription factor regulating <italic>TDH3</italic> expression. (<bold>B</bold>) Summary of gene ontology (GO) enrichment analysis performed with PANTHER tool (<ext-link ext-link-type="uri" xlink:href="http://www.pantherdb.org/">http://www.pantherdb.org/</ext-link>). Fisher’s exact tests were used to evaluate the overrepresentation of GO terms among the 42 genes affected by one or more of the 66 <italic>trans</italic>-regulatory mutations in coding sequences relative to the 1043 genes affected by one or more of the 1251 non-regulatory mutations in coding sequences. The descriptions shown on the left correspond to GO terms with a p-value &lt; 0.05 (left bars), a fold-enrichment &gt; 3 (right bars) and that are not parents to other GO terms in the ontology hierarchy (i.e. GO terms that are the most specific). A more complete list of enriched GO terms can be found in <xref ref-type="supplementary-material" rid="supp8">Supplementary file 8</xref>. Shades of gray represent different categories of GO terms (from darkest to lightest: biological processes, molecular functions and cellular components) or PANTHER pathways (lightest gray). Fold-enrichment was calculated as the observed number of genes with a particular GO term in the set of genes affected by <italic>trans</italic>-regulatory mutations (bold numbers on the right) divided by an expected number of genes obtained from the number of genes with the same GO term in the set of genes affected by non-regulatory mutations (regular numbers on the right). Four groups of GO terms and pathways involved in similar processes are represented by colored areas: chromatin (pink), metabolism (orange), transcription (green), and iron homeostasis (blue).</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-67806-fig6-v1.tif"/></fig><p>At the pathway level, we found that genes involved in glycolysis and de novo purine biosynthesis were also significantly enriched for <italic>trans</italic>-regulatory mutations (<xref ref-type="fig" rid="fig6">Figure 6B</xref>), with the latter driven by the mutations in <italic>ADE2</italic>, <italic>ADE4</italic>, <italic>ADE5,</italic> and <italic>ADE6</italic> genes described above (<xref ref-type="supplementary-material" rid="supp11">Supplementary file 11</xref>). Genes involved in iron homeostasis also emerged as an over-represented group, with five GO terms (including seven genes) being related to the regulation of intracellular iron concentration (<xref ref-type="fig" rid="fig6">Figure 6B</xref>). Diverse cellular processes implicated in iron homeostasis were represented among genes harboring <italic>trans</italic>-regulatory mutations, such as iron transport (<italic>FTR1</italic>, <italic>CCC2</italic>), iron trafficking and maturation of iron-sulfur proteins (<italic>CIA2</italic>, <italic>NAR1</italic>), transcriptional regulation of the iron regulon (<italic>FRA1</italic>), and post-transcriptional regulation of iron homeostasis (<italic>TIS11</italic>). Remarkably, nearly half of all <italic>trans</italic>-regulatory point mutations in coding sequences (31 of 65) were located in genes involved either in purine biosynthesis or iron homeostasis. Moreover, six of the eight genes harboring more than one <italic>trans</italic>-regulatory mutation (<xref ref-type="fig" rid="fig6">Figure 6A</xref>) were involved in one of these two processes. Mutations in purine biosynthesis genes tended to cause large increases in expression, whereas mutations in iron homeostasis genes tended to cause large decreases in expression (<xref ref-type="supplementary-material" rid="supp11">Supplementary file 11</xref>). Although the mechanistic relationship between these pathways and <italic>TDH3</italic> expression is not known, changing cellular conditions, including concentrations of metabolites (<xref ref-type="bibr" rid="bib63">Pinson et al., 2009</xref>) or iron within the cell (reviewed in <xref ref-type="bibr" rid="bib59">Outten and Albetel, 2013</xref>), can affect the regulation of gene expression. Ultimately, our data suggest that although mutations affecting <italic>P<sub>TDH3</sub>-YFP</italic> expression map to genes with diverse functions, genes involved in a small number of well-defined biological processes are particularly likely to harbor such <italic>trans</italic>-regulatory mutations.</p></sec><sec id="s2-8"><title><italic>Trans</italic>-regulatory mutations are enriched in genomic regions harboring natural variation affecting <italic>TDH3</italic> expression</title><p>Because new mutations affecting gene expression provide the raw material for regulatory variation segregating within a species, we asked whether the <italic>trans</italic>-regulatory mutations we observed were enriched in genomic regions associated with naturally occurring <italic>trans</italic>-regulatory variation affecting expression driven by the <italic>TDH3</italic> promoter. Specifically, we compared the genomic locations of <italic>trans</italic>-regulatory mutations identified in the current study to the locations of <italic>trans</italic>-acting quantitative trait loci (QTL) affecting expression of <italic>P<sub>TDH3</sub>-YFP</italic> identified from crosses between the progenitor strain of the EMS mutants (BY) and three other <italic>S. cerevisiae</italic> strains (SK1, YPS1000, M22) (<xref ref-type="bibr" rid="bib54">Metzger and Wittkopp, 2019</xref>; <xref ref-type="fig" rid="fig7">Figure 7A</xref>).</p><fig-group><fig id="fig7" position="float"><label>Figure 7.</label><caption><title>Overrepresentation of <italic>trans</italic>-regulatory mutations in eQTLs regions.</title><p>(<bold>A</bold>) Overlap of 66 <italic>trans</italic>-regulatory point mutations and 317 eQTL regions along the yeast genome. eQTL regions were identified by BSA-Seq in <xref ref-type="bibr" rid="bib54">Metzger and Wittkopp, 2019</xref> from three crosses of a laboratory strain (BY) to each of three strains expressing <italic>P<sub>TDH3</sub>-YFP</italic> in the genetic background of different <italic>S. cerevisiae</italic> isolates: SK1 (eQTL regions represented by blue bars), YPS1000 (eQTL regions represented by yellow bars) and M22 (eQTL regions represented by red bars). Triangles indicate the genomic locations of <italic>trans</italic>-regulatory mutations, with open triangles representing mutations identified in mutants from the unenriched collection and filled triangles representing mutations identified in mutants enriched for large effects. Triangles are colored depending on the overlap between mutations and eQTL regions: black if the mutation is outside of any eQTL region, blue if the mutation lies in an eQTL region only identified from SK1xBY, yellow if the mutation lies in an eQTL region only identified from YPS1000xBY, red if the mutation lies in an eQTL region only identified from M22xBY, green if the mutation lies in two overlapping eQTL regions identified from SK1xBY and YPS1000xBY, purple if the mutation lies in two overlapping eQTL regions identified from SK1xBY and M22xBY, orange if the mutation lies in two overlapping eQTL regions identified from M22xBY and YPS1000xBY and brown if the mutation lies in three overlapping eQTL regions identified from the three crosses. (<bold>B</bold>) Proportions of non-regulatory and <italic>trans</italic>-regulatory mutations located in eQTL regions. Black bars: proportions of sites among the 12.07 Mb yeast genome. Blue bars: proportions of the 1759 non-regulatory point mutations. Orange bars: proportions of the 66 <italic>trans</italic>-regulatory mutations (excluding aneuploidies). Red bars: proportions of the 44 <italic>trans</italic>-regulatory mutations identified in mutants from the unenriched collection. Green bars: proportions of the 22 <italic>trans</italic>-regulatory mutations identified in mutants enriched for large effects. The proportions of non-regulatory and <italic>trans</italic>-regulatory mutations in eQTL regions were compared using <italic>G</italic>-tests (***: p &lt; 0.001, **: 0.001 &lt; p &lt; 0.01, *: 0.01 &lt; p &lt; 0.05, ns: p &gt; 0.05).</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-67806-fig7-v1.tif"/></fig><fig id="fig7s1" position="float" specific-use="child-fig"><label>Figure 7—figure supplement 1.</label><caption><title>Proportions of different categories of non-regulatory mutations and <italic>trans</italic>-regulatory mutations located in eQTLs regions.</title><p>Black bars: proportions of all sites among the 12.07 Mb yeast genome. Medium blue bars: proportions of the 1759 non-regulatory point mutations. Light blue bars: proportions of non-regulatory mutations at sites for which the total sequencing depth was below the median sequencing depth of the corresponding library in BSA-Seq data. Dark blue bars: proportions of non-regulatory mutations at sites for which the total sequencing depth was equal or above the median sequencing depth of the corresponding library in BSA-Seq data. Orange bars: proportions of the 66 <italic>trans</italic>-regulatory mutations (excluding aneuploidies). Red bars: proportions of the 49 <italic>trans</italic>-regulatory mutations identified by BSA-Seq. Green bars: proportions of the 17 <italic>trans</italic>-regulatory mutations identified by Sanger sequencing of candidate genes. The proportions of non-regulatory and <italic>trans</italic>-regulatory mutations in eQTL regions were compared using <italic>G</italic>-tests (***: p &lt; 0.001, **: 0.001 &lt; p &lt; 0.01, *: 0.01 &lt; p &lt; 0.05, ns: p &gt; 0.05).</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-67806-fig7-figsupp1-v1.tif"/></fig></fig-group><p>Non-regulatory mutations were observed in eQTL regions as often as expected by chance (66.7% of non-regulatory mutations <italic>vs</italic> 65.1% of the whole genome in eQTL regions; <italic>G</italic>-test: p = 0.15), but the 66 <italic>trans-</italic>regulatory mutations were significantly enriched in eQTL regions (<xref ref-type="fig" rid="fig7">Figure 7B</xref>; 88% of <italic>trans-</italic>regulatory mutations <italic>vs</italic> 66.7% of non-regulatory mutations in eQTL regions; <italic>G</italic>-test: p = 9.6 x 10<sup>−5</sup>). The overrepresentation of <italic>trans-</italic>regulatory mutations in eQTL regions remained statistically significant when we considered only the 44 <italic>trans-</italic>regulatory mutations identified from the collection of EMS mutants not enriched for large effects (<xref ref-type="fig" rid="fig7">Figure 7B</xref>; <italic>G</italic>-test: p = 0.027) or when we excluded the 17 <italic>trans-</italic>regulatory mutations identified by sequencing candidate genes (<xref ref-type="fig" rid="fig7s1">Figure 7—figure supplement 1</xref>; <italic>G</italic>-test: p = 8.4 x 10<sup>−3</sup>). The enrichment of <italic>trans-</italic>regulatory mutations in eQTL regions was thus not driven solely by the effect size of these mutations or by the fact that several of the <italic>trans-</italic>regulatory mutations with large effects were located in the same genes. We also found that differences in sequencing coverage across the genome were unlikely to account for this enrichment (<xref ref-type="fig" rid="fig7s1">Figure 7—figure supplement 1</xref>). When we considered eQTL regions identified from each cross separately, we observed a significant enrichment of <italic>trans</italic>-regulatory mutations in eQTL regions identified in SK1 x BY and YPS1000 x BY crosses, but not in eQTL regions identified in the M22 x BY cross (<xref ref-type="fig" rid="fig7">Figure 7B</xref>; <italic>G</italic>-tests: p = 0.016 for SK1 x BY, p = 6.5 x 10<sup>−3</sup> for YPS1000 x BY, p = 0.70 for M22 x BY). Overall, the enrichment of <italic>trans</italic>-regulatory mutations in eQTL regions suggests that biases in the mutational sources of regulatory variation have shaped genetic sources of expression variation segregating in wild populations.</p></sec></sec><sec id="s3" sec-type="discussion"><title>Discussion</title><p>By systematically isolating and characterizing 69 <italic>trans</italic>-regulatory mutations that all affect expression of the same focal gene, this study reveals how <italic>trans</italic>-regulatory mutations are distributed within a genome and within a regulatory network. For example, we found that these <italic>trans</italic>-regulatory mutations were widely spread throughout the genome, with all except one located in coding sequences. These data also allowed us to determine how well a regulatory network inferred from integrating functional genomic and genetic data can predict sources of <italic>trans</italic>-regulatory variation. Like many biological networks, transcriptional regulatory networks have been inferred with the promise of explaining relationships between genetic variants and the higher order trait of gene expression, but the predictive power of such networks remains sparsely tested (<xref ref-type="bibr" rid="bib14">Flint and Ideker, 2019</xref>).</p><p>We found that although the <italic>trans</italic>-regulatory mutations in coding regions were not enriched in transcription factors generally, they were overrepresented among transcription factors inferred to be regulators of <italic>TDH3</italic>. None of these transcription factors are known to directly bind to the <italic>TDH3</italic> promoter, however, and mutations in <italic>RAP1</italic> and <italic>GCR1,</italic> which have well characterized binding sites in the <italic>TDH3</italic> promoter, were notably missing from our set of <italic>trans</italic>-regulatory mutations affecting <italic>P<sub>TDH3</sub>-YFP</italic> expression. Targeted mutagenesis of <italic>RAP1</italic> and <italic>GCR1</italic> suggested that most mutations in these genes (particularly <italic>RAP1</italic>) cause severe growth defects that might have prevented their recovery in mutagenesis screens. Over 90% of the <italic>trans</italic>-regulatory mutations examined were located in genes outside of this transcription factor network encoding proteins with diverse molecular functions involved in chromatin remodeling, nonsense-mediated mRNA decay, translation regulation, purine biosynthesis, iron homeostasis, and glucose sensing. Surprisingly, nearly half of the <italic>trans</italic>-regulatory mutations mapped to genes involved in either the purine biosynthesis or iron homeostasis pathways. Although not anticipated, finding so many <italic>trans</italic>-regulatory mutations in genes that are not transcription factors is consistent with the transcriptomic effects of gene deletions showing that transcription factors tend not to affect expression of more genes than other types of proteins (<xref ref-type="bibr" rid="bib12">Featherstone and Broadie, 2002</xref>). Consequently, it seems that regulatory networks describing the relationships between transcription factors and target genes might capture only a small fraction of the potential sources of <italic>trans</italic>-regulatory variation.</p><p>Understanding the properties of <italic>trans</italic>-regulatory mutations is important because these mutations provide the raw material for natural <italic>trans</italic>-regulatory variation. We found that mutations affecting <italic>P<sub>TDH3</sub>-YFP</italic> expression were enriched in genomic regions associated with expression variation among wild isolates of <italic>S. cerevisiae</italic>, suggesting that mutational sources of regulatory variation have had a lasting effect on sources of genetic variation affecting gene expression segregating in natural populations. This pattern is not necessarily expected if the <italic>trans</italic>-regulatory mutations we characterized captured only a small subset of the loci that can contribute to segregating <italic>trans</italic>-regulatory variation for this gene. Differences between the distribution of new <italic>trans</italic>-regulatory mutations and segregating <italic>trans</italic>-regulatory variants are also expected to arise when natural selection favors the maintenance of mutations at some loci more than others. Such differences in fitness can arise independently of a mutation’s impact on <italic>TDH3</italic> expression because <italic>trans</italic>-acting mutations can also have pleiotropic effects on expression of other genes. A third reason why differences between the mutational sources of <italic>trans</italic>-regulatory variation characterized here and <italic>trans-</italic>regulatory variation segregating in the wild can occur would be because of epistatic interactions among variants that are not captured by studying the effects of mutations individually. Ultimately, explaining the variation in gene expression we see in natural populations will require studies like this elucidating the mutational input as well as studies describing the fitness, pleiotropic, and epistatic effects of these mutations in native environments.</p><p>To the best of our knowledge this work provides the largest collection of individual mutations with <italic>trans</italic>-regulatory effects on expression of a single gene available to date, but it still only interrogates a single gene in a single species. Moreover, although the methods used were sensitive enough to identify genetic changes impacting expression of the focal gene as little as 1.6%, many mutations important for natural variation might have even smaller individual effects on a focal gene’s expression and are thus missing from this study (<xref ref-type="bibr" rid="bib65">Rockman, 2012</xref>). The chemical mutagen (EMS) used to generate the mutants analyzed in this work also captures only a subset of the type of mutations that arise naturally, and the use of a YFP reporter gene to measure activity of the <italic>TDH3</italic> promoter precluded recovery of <italic>trans</italic>-regulatory mutations that can impact native <italic>TDH3</italic> expression post-transcriptionally. The focal gene chosen for this work, <italic>TDH3</italic>, might also have properties that cause its spectrum of <italic>trans</italic>-regulatory mutations to differ from other genes in <italic>S. cerevisiae</italic>. For example, <italic>TDH3</italic> is one of the most highly expressed genes in <italic>S. cerevisiae</italic> (<xref ref-type="bibr" rid="bib18">Ghaemmaghami et al., 2003</xref>), and it is one of the ~8% of genes in the <italic>S. cerevisiae</italic> genome that contains both a TATA box and a large nucleosome-free region in its promoter (<xref ref-type="bibr" rid="bib77">Tirosh and Barkai, 2008</xref>). The metabolic functions of the TDH3p protein encoded by the <italic>TDH3</italic> gene might also cause its regulatory network to have properties that differ from genes encoding proteins with other types of functions (<xref ref-type="bibr" rid="bib44">Luscombe et al., 2004</xref>).</p><p>It is tempting to extend these results from <italic>S. cerevisiae</italic> to other eukaryotes, but such extrapolation must take into account differences in genomes and gene regulatory mechanisms among species. For example, compared to species like fruit flies, mice, and humans, the baker’s yeast <italic>S. cerevisiae</italic> has a much higher proportion of its genome (69.4%, <ext-link ext-link-type="uri" xlink:href="https://www.yeastgenome.org/">https://www.yeastgenome.org/</ext-link>) that codes for proteins and much more compact <italic>cis</italic>-regulatory sequences (the median promoter length is 455 bp <xref ref-type="bibr" rid="bib35">Kristiansson et al., 2009</xref>). Consequently, new <italic>trans</italic>-regulatory mutations in coding sequences might be more likely to arise in <italic>S. cerevisiae</italic> than in these other species. Most <italic>S. cerevisiae</italic> genes also lack introns (<xref ref-type="bibr" rid="bib60">Parenteau et al., 2019</xref>) and DNA methylation is less prevalent than in many other eukaryotic species (<xref ref-type="bibr" rid="bib74">Tang et al., 2012</xref>), so these potential sources of <italic>trans-</italic>regulatory variation in other species are unlikely to be captured when studying regulatory mutations in <italic>S. cerevisiae</italic>. Nonetheless, we think some observations, such as that genes with diverse functions can harbor <italic>trans</italic>-regulatory mutations, are likely to also apply to other eukaryotic species. Ultimately, we believe that this work provides an important foundation for understanding how the <italic>trans</italic>-regulatory mutations that give rise to <italic>trans</italic>-regulatory variation segregating in natural populations are structured within a genome and a regulatory network.</p></sec><sec id="s4" sec-type="materials|methods"><title>Materials and methods</title><sec id="s4-1"><title>Mutant strains selected for mapping</title><p>To identify mutations associated with expression changes, we selected 82 haploid mutant strains for bulk segregant analysis (<xref ref-type="fig" rid="fig1">Figure 1A</xref>) from three collections of mutants obtained in <xref ref-type="bibr" rid="bib20">Gruber et al., 2012</xref> and <xref ref-type="bibr" rid="bib53">Metzger et al., 2016</xref> via ethyl methanesulfonate (EMS) mutagenesis of two progenitor strains expressing a <italic>YFP</italic> reporter gene (Yellow Fluorescent Protein) under control of the <italic>TDH3</italic> promoter (<italic>P<sub>TDH3</sub>-YFP</italic>). 71 mutants were selected from a collection of 1498 lines founded from cells isolated randomly (unenriched) after mutagenesis in <xref ref-type="bibr" rid="bib53">Metzger et al., 2016</xref>, five mutants were selected from 211 lines founded from cells enriched for fluorescence changes after mutagenesis in <xref ref-type="bibr" rid="bib53">Metzger et al., 2016</xref> and the last six mutants were selected from 1064 lines founded from cells enriched for fluorescence changes in <xref ref-type="bibr" rid="bib20">Gruber et al., 2012</xref>. Mutants from <xref ref-type="bibr" rid="bib53">Metzger et al., 2016</xref> were obtained by mutagenesis of the progenitor strain YPW1139 (<italic>MATα ura3d0</italic>), while mutants from <xref ref-type="bibr" rid="bib20">Gruber et al., 2012</xref> were obtained by mutagenesis of the progenitor strain YPW1 (<italic>MAT</italic><bold>a</bold> <italic>ura3d0 lys2d0</italic>). Both progenitors were derived from S288c genetic background (see <xref ref-type="bibr" rid="bib53">Metzger et al., 2016</xref> and <xref ref-type="bibr" rid="bib20">Gruber et al., 2012</xref> for details on construction of YPW1139 and YPW1 strains). In YPW1139, <italic>P<sub>TDH3</sub>-YFP</italic> is inserted at the <italic>ho</italic> locus with a <italic>KanMX</italic> drug resistance marker. In YPW1, <italic>P<sub>TDH3</sub>-YFP</italic> is inserted at position 199270 on chromosome I near a pseudogene. YPW1139 harbors <italic>RME1(ins-308A)</italic> and <italic>TAO3(1493Q)</italic> alleles (<xref ref-type="bibr" rid="bib8">Deutschbauer and Davis, 2005</xref>) that increase sporulation frequency relative to YPW1 alleles, as well as <italic>SAL1</italic>, <italic>CAT5</italic> and <italic>MIP1</italic> alleles that decrease the frequency of the petite phenotype (<xref ref-type="bibr" rid="bib9">Dimitrov et al., 2009</xref>). We previously showed that the few genetic differences between YPW1 and YPW1139 did not affect the magnitude of effects of <italic>TDH3</italic> promoter mutations on fluorescence (<xref ref-type="bibr" rid="bib53">Metzger et al., 2016</xref>). Fluorescence levels of the three collections were measured in <xref ref-type="bibr" rid="bib20">Gruber et al., 2012</xref> and in <xref ref-type="bibr" rid="bib53">Metzger et al., 2016</xref>. From these data, we selected 39 mutants for BSA-Seq that showed statistically significant fluorescence changes greater than 1% relative to the progenitor strain. Among these mutants, six were selected from the <xref ref-type="bibr" rid="bib20">Gruber et al., 2012</xref> collection (<italic>Z</italic>-score &gt; 2.58, p &lt; 0.01), five were selected from mutants enriched for large effects in <xref ref-type="bibr" rid="bib53">Metzger et al., 2016</xref> (permutation test, p &lt; 0.05) and 28 were selected from unenriched mutants in <xref ref-type="bibr" rid="bib53">Metzger et al., 2016</xref> (permutation test, p &lt; 0.05). The remaining 43 mutants included in BSA-Seq experiments were selected from mutants in <xref ref-type="bibr" rid="bib53">Metzger et al., 2016</xref> for which we collected new fluorescence measures using flow cytometry. This second fluorescence screen included 197 lines from the unenriched collection that were chosen because they showed statistically significant fluorescence changes (permutation test, p &lt; 0.05) greater than 1% relative to the progenitor strain in the initial screen published in <xref ref-type="bibr" rid="bib53">Metzger et al., 2016</xref>. The 43 mutants selected from this 2nd screen showed statistically significant fluorescence changes (permutation test, p &lt; 0.05) greater than 1% relative to the progenitor strain.</p></sec><sec id="s4-2"><title>Measuring YFP expression by flow cytometry</title><p>Fluorescence levels of mutant strains were quantified by flow cytometry using the same approach as described in <xref ref-type="bibr" rid="bib53">Metzger et al., 2016</xref> and <xref ref-type="bibr" rid="bib11">Duveau et al., 2018</xref>. For assays involving strains stored in individual tubes at −80°C, all strains were thawed in parallel on YPG plates (10 g yeast extract, 20 g peptone, 50 ml glycerol, 20 g agar per liter) and grown for 2 days at 30°C. Strains were then arrayed using pipette tips in 96 deep well plates containing 0.5 ml of YPD medium (10 g yeast extract, 20 g peptone, 20 g D-glucose per liter) per well at positions defined in <xref ref-type="supplementary-material" rid="supp12">Supplementary file 12</xref>. The reference strain YPW1139 was inoculated at 20 fixed positions on each plate to correct for plate and position effects on fluorescence. The non-fluorescent strain YPW978 was inoculated in one well per plate to quantify the autofluorescence of yeast cells. Plates were incubated at 30°C for 20 hr with 250 rpm orbital shaking (each well contained a sterile 3 mm glass bead to maintain cells in suspension). Samples from each plate were then transferred to omnitrays containing YPG-agar using a V&amp;P Scientific pin tool. For assays involving strains already arrayed in 96-well plates at −80°C (i.e<italic>. RAP1</italic> and <italic>GCR1</italic> mutants), strains were directly transferred on YPG omnitrays after thawing. After 48 hr of incubation at 30°C, samples from each omnitray were inoculated using the pin tool in four replicate 96-well plates containing 0.5 ml of YPD per well and cultivated at 30°C with 250 rpm shaking for 22 hr. Then, 15 µl of cell cultures were transferred to a 96-well plate with 0.5 ml of PBS per well (phosphate-buffered saline) and samples were immediately analyzed on a BD Accuri C6 flow cytometer connected to a HyperCyt autosampler (IntelliCyt Corp). A 488 nm laser was used for excitation and the YFP signal was acquired with a 530/30 optical filter. Each well was sampled for 2 s, yielding fluorescence and cell size measurements for at least 5000 events per well. Flow cytometry data were analyzed using custom R scripts (<xref ref-type="supplementary-material" rid="scode1">Source code 1</xref>) as described in <xref ref-type="bibr" rid="bib11">Duveau et al., 2018</xref>. First, events that did not correspond to single cells were filtered out using <italic>flowClust</italic> clustering functions. Second, fluorescence intensity was scaled by cell size in several steps. For <xref ref-type="fig" rid="fig1">Figure 1B–D</xref>, these values of fluorescence relative to cell size were directly used for subsequent steps of the analysis. For other figures, these values were transformed using a log-linear function to be linearly related with YFP abundance. Transformations of fluorescence values were performed using the relationship between fluorescence levels and <italic>YFP</italic> mRNA levels established in <xref ref-type="bibr" rid="bib11">Duveau et al., 2018</xref> from five strains carrying mutations in the promoter of the <italic>P<sub>TDH3</sub>-YFP</italic> reporter gene. The YFP mRNA levels quantified in these five strains are expected to be linearly related with YFP protein abundance based on a previous study that compared mRNA and protein levels for a similar fluorescent protein (GFP) across a broad range of expression levels (<xref ref-type="bibr" rid="bib31">Kafri et al., 2016</xref>). For this reason and because mutations recovered in this study may alter YFP expression at the post-transcriptional level, the transformed values of fluorescence were considered to provide estimates of YFP abundance instead of mRNA levels. The median expression among all cells of each sample was then corrected to account for positional effects estimated from a linear model applied to the median expression of the 20 control samples on each plate. To correct for autofluorescence, the mean of median expression measured among all replicate populations of the non-fluorescent strain was then subtracted from the median expression of each sample. Finally, a relative measure of expression was calculated by dividing the median expression of each sample by the mean of the median expression among replicates of the reference strain. Samples for which the relative expression differed from the median expression among replicate populations by more than five times the median absolute deviation measured among replicate populations were considered as outliers and ignored. Figures show the mean relative expression among replicate populations of each genotype. Permutation tests used to compare the expression level of each single site mutant to the expression level of the EMS mutant carrying the same mutation are described in the legend of <xref ref-type="fig" rid="fig2s5">Figure 2—figure supplement 5A</xref>.</p></sec><sec id="s4-3"><title>Two-level permutation tests</title><p>We developed a permutation-based approach to determine which EMS mutant strains from <xref ref-type="bibr" rid="bib53">Metzger et al., 2016</xref> showed a significant change in YFP expression relative to their progenitor strain. This permutation approach was motivated by the fact that Student tests and Mann-Whitney-Wilcoxon tests applied to these data appeared to be overpowered. Indeed, the flow cytometry assay from <xref ref-type="bibr" rid="bib53">Metzger et al., 2016</xref> included 146 instances of the progenitor strain YPW1139 that were placed at random plate positions and with fluorescence measured in four replicate populations for each position. When comparing the mean expression of the four replicate populations of YPW1139 grown at a given plate position to the mean expression of all other replicate populations of YPW1139, the p-value was below 0.05 in 25.3% of cases when using Student tests and in 13.7% of cases when using Mann-Whitney-Wilcoxon tests. The fact that more than 5% of p-values were below 0.05 indicated that the tests were overpowered, which was because expression differences between YPW1139 populations grown at different plate positions were in average larger than expression differences between replicate populations grown at the same position. For this reason, we compared the expression of each mutant strain to the expression of the 146 x 4 populations of the YPW1139 progenitor strain using permutation tests with two levels of resampling as described below. In these tests, we compared 10,000 times the expression levels of each tested strain measured in quadruplicates to the expression levels of YPW1139 measured in quadruplicates at a randomly selected plate position among the 146 available positions (a new position was picked at each iteration). For each iteration of the comparison, we calculated the difference <italic>D</italic> between (1) the absolute difference observed between the mean expression of the tested strain and the mean expression of YPW1139 and (2) a randomized absolute difference of mean expression between two sets of four expression values obtained by random permutation of the four expression values measured for the tested strain and of the four expression values measured for YPW1139 at the selected plate position. Finally, for each tested strain the proportion of <italic>D</italic> values that were negative (after excluding <italic>D</italic> values equal to zero) corresponded to the p-value of the permutation test. When we applied this test to YPW1139 as a tested strain, we found that the p-value was below 0.05 for 6.1% of the 146 plate positions containing YPW1139, indicating that the permutation test was not overpowered.</p></sec><sec id="s4-4"><title>BSA-Seq procedure</title><p>To identify mutations associated with fluorescence levels in EMS-treated mutants, we used bulk-segregant analysis followed by Illumina sequencing (BSA-Seq). BSA-Seq data corresponding to the six mutants from <xref ref-type="bibr" rid="bib20">Gruber et al., 2012</xref> were collected together with the BSA-Seq dataset published in <xref ref-type="bibr" rid="bib10">Duveau et al., 2014</xref>. For the other 76 mutants (from <xref ref-type="bibr" rid="bib53">Metzger et al., 2016</xref>), BSA-Seq data were collected in this study in several batches (see <xref ref-type="supplementary-material" rid="supp13">Supplementary file 13</xref>) using the experimental approach described in <xref ref-type="bibr" rid="bib10">Duveau et al., 2014</xref> (with few modifications). First, each EMS-treated mutant (<italic>MATα ura3d0 ho::P<sub>TDH3</sub>-YFP ho::KanMX</italic>) was crossed to the mapping strain YPW1240 (<italic>MAT</italic><bold>a</bold> <italic>ura3d0 ho::P<sub>TDH3</sub>-YFP ho::NatMX4 mata2::yEmRFP-HygMX</italic>) that contained the <italic>FASTER MT</italic> system from <xref ref-type="bibr" rid="bib5">Chin et al., 2012</xref> used to tag diploid and <italic>MAT</italic><bold>a</bold> cells with a fluorescent reporter. Crosses were performed on YPD agar plates and replica-plated on YPD + G418 + Nat medium (YPD agar with 350 mg/L geneticin (G418) and 100 mg/L Nourseothricin) to select diploid hybrids. After growth, cells were streaked on another YPD + G418 + Nat agar plate, one colony was patched on YPG agar for each mutant and the diploid strain was kept frozen at −80°C. Bulk segregant populations were then collected for batches of eight mutants in parallel as follows. Diploid strains were thawed and revived on YPG plates, grown for 12 hr at 30°C on GNA plates (50 g D-glucose, 30 g Difco nutrient broth, 10 g yeast extract and 20 g agar per liter) and sporulation was induced for 4 days at room temperature on KAc plates (10 g potassium acetate and 20 g agar per liter). For each mutant, we then isolated a large population of random spores (&gt; 10<sup>8</sup> spores) by digesting tetrads with zymolyase, vortexing, and sonicating samples in 0.02% triton-X (exactly as described in <xref ref-type="bibr" rid="bib10">Duveau et al., 2014</xref>). ~3 x 10<sup>5</sup> <italic>MATα</italic> spores were sorted by FACS (BD FACSAria II) based on the absence of RFP fluorescence signal measured using a 561 nm laser and 582/15 optical filter. Spores were then resuspended in 2 ml of YPD medium. After 24 hr of growth at 30°C, 0.4 ml of cell culture was transferred to a 5 ml tube containing 2 ml of PBS. Three populations of 1.5 x 10<sup>5</sup> segregant cells were then collected by FACS: (1) a low fluorescence population of cells sorted among the 2.5% of cells with lowest fluorescence levels (‘low bulk’), (2) a high fluorescence population of cells sorted among the 2.5% of cells with highest fluorescence levels (‘high bulk’), and (3) a control population of cells sorted regardless of their fluorescence levels. YFP signal was measured using a 488 nm laser and a 530/30 optical filter. To exclude budding cells and enrich for single cells, ~70% of all events were filtered out based on the area and width of the forward scatter signal prior to sorting. In addition, the median FSC.A (area of forward scatter, a proxy for cell size) was maintained to similar values in the low fluorescence bulk and in the high fluorescence bulk by drawing sorting gates that were parallel to the linear relationship between FSC.A and fluorescence intensity in the FACSDiva software. After sorting, cells were resuspended in 1.6 ml of YPD medium and grown for 30 hr at 30°C. Each sample was then stored at −80°C in 15% glycerol in two separate tubes: one tube containing 1 ml of culture (for DNA extraction) and one tube containing 0.5 ml of culture (for long-term storage). Extraction of genomic DNA was performed for 24 samples in parallel using a Gentra Puregene Yeast/Bact kit (Qiagen). Then, DNA libraries were prepared from 1 ng of genomic DNA using Nextera XT DNA Library Prep kits (Illumina) for low fluorescence bulks and for high fluorescence bulks (control populations were not sequenced). Tagmentation was carried out at 55°C for 5 min. Dual indexing of the libraries was achieved using index adapters provided in the Nextera XT Index kit (index sequences used for each library are indicated in <xref ref-type="supplementary-material" rid="supp14">Supplementary file 14</xref>). Final library purification and size selection was achieved using Agencourt AMPure XP beads (30 µl of beads added to 50 µl of PCR-amplified libraries followed by ethanol washes and resuspension in 50 µl of Tris-EDTA buffer). The average size of DNA fragments in the final libraries was 650 bp, as quantified from a subset of samples using high sensitivity assays on a 2100 Bioanalyzer (Agilent). The concentration of all libraries was quantified with a Qubit 2.0 Fluorometer (Thermo Fisher Scientific) using dsDNA high sensitivity assays. Libraries to be sequenced in the same flow lane were pooled to equal concentration in a single tube and sequenced on a HiSeq4000 instrument (Illumina) at the University of Michigan Sequencing Core Facility (150 bp paired-end sequencing). The 2 x 76 libraries were sequenced in four distinct sequencing runs (45300, 45301, 54374 and 54375) that included 36–54 samples (libraries sequenced in each run are indicated in <xref ref-type="supplementary-material" rid="supp14">Supplementary file 14</xref>). In addition, four control libraries were sequenced in run 45300, corresponding to genomic DNA from (1) YPW1139 progenitor strain, (2) YPW1240 mapping strain, (3) a bulk of low fluorescence segregants from YPW1139 x YPW1240 cross, and (4) a bulk of high fluorescence segregants from YPW1139 x YPW1240 cross. 18 libraries sequenced in run 54374 were not analyzed in this study.</p></sec><sec id="s4-5"><title>Analysis of BSA-Seq data</title><p>Demultiplexing of sequencing reads and generation of FASTQ files were performed using Illumina <italic>bcl2fastq</italic> v1.8.4 for sequencing runs 45300 and 45301 and <italic>bcl2fastq2</italic> v2.17 for runs 54374 and 54375. The next steps of the analysis were processed on the Flux cluster administered by the Advanced Research Computing Technology Services of the University of Michigan (script available in <xref ref-type="supplementary-material" rid="scode4">Source code 4</xref>). First, low quality ends of reads were trimmed with <italic>sickle</italic> (<ext-link ext-link-type="uri" xlink:href="https://github.com/najoshi/sickle">https://github.com/najoshi/sickle</ext-link>; <xref ref-type="bibr" rid="bib30">Joshi and Fass, 2011</xref>) and adapter sequences were removed with <italic>cutadapt</italic> (<xref ref-type="bibr" rid="bib47">Martin, 2011</xref>). Reads were then aligned to the S288c reference genome (<ext-link ext-link-type="uri" xlink:href="https://www.yeastgenome.org/">https://www.yeastgenome.org/</ext-link>, R64-1-1 release to which we added the sequences corresponding to <italic>P<sub>TDH3</sub>-YFP</italic>, <italic>KanMX</italic>, and <italic>NatMX4</italic> transgenes, available in <xref ref-type="supplementary-material" rid="supp12">Supplementary file 12</xref>) using <italic>bowtie2</italic> (<xref ref-type="bibr" rid="bib37">Langmead and Salzberg, 2012</xref>) and overlaps between paired reads were clipped using <italic>clipOverlap</italic> in <italic>bamUtil</italic> (<ext-link ext-link-type="uri" xlink:href="https://github.com/statgen/bamUtil">https://github.com/statgen/bamUtil</ext-link>). The sequencing depth at each position in the genome was determined using <italic>bedtools genomecov</italic> (<ext-link ext-link-type="uri" xlink:href="https://github.com/arq5x/bedtools2">https://github.com/arq5x/bedtools2</ext-link>). For variant calling, BAM files corresponding to the low fluorescence bulk and to the high fluorescence bulk of each mutant were processed together using <italic>freebayes</italic> (<ext-link ext-link-type="uri" xlink:href="https://github.com/ekg/freebayes">https://github.com/ekg/freebayes</ext-link>; <xref ref-type="bibr" rid="bib16">Garrison and Marth, 2012</xref>) with options <italic><monospace>--pooled-discrete</monospace> <monospace>--pooled-continuous</monospace></italic>. That way, sequencing data from both bulks were pooled to increase the sensitivity of variant calling and allele counts were reported separately for each bulk. To obtain a list of mutations present in each mutant strain, false positive calls in the VCF files generated by <italic>freebayes</italic> were then filtered out with the Bioconductor package <italic>VariantAnnotation</italic> in R (<xref ref-type="supplementary-material" rid="scode2">Source code 2</xref>). Filtering was based on the values of several parameters such as quality of genotype inference (QUAL &gt; 200), mapping quality (MQM &gt; 27), sequencing depth (DP &gt; 20), counts of reference and alternate alleles (AO &gt; three and RO &gt; 3), frequency of the reference allele (FREQ.REF &gt; 0.1), proportion of reference and alternate alleles supported by properly paired reads (PAIRED &gt; 0.8 and PAIREDR &gt; 0.8), probability to observe the alternate allele on both strands (SAP &lt; 100) and at different positions of the reads (EPP &lt; 50 and RPP &lt; 50). The values of these parameters were chosen to filter out a maximum number of calls while retaining 28 variants previously confirmed by Sanger sequencing. We then used likelihood ratio tests (<italic>G</italic>-tests) in R to determine for each variant site whether the frequency of the alternate allele (i.e. the mutation) was statistically different between the low fluorescence bulk and the high fluorescence bulk (<xref ref-type="supplementary-material" rid="scode2">Source code 2</xref>). A point mutation was considered to be associated with fluorescence (directly or by linkage) if the p-value of the <italic>G</italic>-test was below 0.001, corresponding to a <italic>G</italic> value above 10.828. Since this <italic>G</italic>-test was performed for a total of 1819 mutations, we expected that 1.82 mutations would be associated with fluorescence due to type I error (false positives) at a p-value threshold of 0.001. This expected number of false positives was considered acceptable since it represented only 2.7% of all mutations that were associated with fluorescence. To determine if an aneuploidy was associated with fluorescence level, we compared the sequencing coverage of the aneuploid chromosome to genome-wide sequencing coverage in the low and high fluorescence bulks using <italic>G-</italic>tests. The <italic>G</italic> statistics was computed from the number of reads mapping to the aneuploid chromosome and the number of reads mapping to the rest of the genome in the low and high fluorescence bulks. Aneuploidies with <italic>G</italic> &gt; 10.828, which corresponds to p-value &lt; 0.001, were considered to be present at statistically different frequencies in both bulks. A custom R script was used to annotate all mutations identified in BSA-Seq data (<xref ref-type="supplementary-material" rid="scode3">Source code 3</xref>), retrieving information about the location of mutations in intergenic, intronic or exonic regions, the name of genes affected by coding mutations or the name of neighboring genes in case of intergenic mutations and the expected impact on amino acid sequences (synonymous, nonsynonymous, or nonsense mutation and identity of the new amino acid in case of a nonsynonymous mutation).</p></sec><sec id="s4-6"><title>Sanger sequencing of candidate genes</title><p>As an alternative approach to BSA-Seq, additional mutations were identified by directly sequencing candidate genes in a subset of EMS-treated mutants (<xref ref-type="supplementary-material" rid="supp4">Supplementary file 4</xref>). More specifically, we sequenced the <italic>P<sub>TDH3</sub>-YFP</italic> transgene in 95 mutant strains from <xref ref-type="bibr" rid="bib53">Metzger et al., 2016</xref> that showed decreased fluorescence by more than 10% relative to the progenitor strain. We sequenced the <italic>ADE4</italic> coding sequence in 14 mutants from <xref ref-type="bibr" rid="bib53">Metzger et al., 2016</xref> that were not included in the BSA-Seq assays and that showed increased fluorescence by more than 5% relative to the progenitor strain. Two of the sequenced mutants had a mutation in the <italic>ADE4</italic> coding sequence. We then sequenced the <italic>ADE5</italic> coding sequence in the remaining 12 mutants and found a mutation in five of the sequenced mutants. We continued by sequencing the <italic>ADE6</italic> coding sequence in the remaining seven mutants. Five of the sequenced mutants had a single mutation and one mutant had two mutations in the <italic>ADE6</italic> coding sequence. We sequenced the <italic>ADE8</italic> coding sequence in the last mutant but we found no candidate mutation in this mutant. Finally, we sequenced the <italic>ADE2</italic> coding sequence in two mutants that showed a reddish color when growing on YPD plates. For all genes, the sequenced region was amplified by PCR from cell lysates, PCR products were cleaned up using Exo-AP treatment (7.5 µl PCR product mixed with 0.5 µl Exonuclease-I (NEB), 0.5 µl Antarctic Phosphatase (NEB), 1 µl Antarctic Phosphastase buffer and 0.5 µl H<sub>2</sub>O incubated at 37°C for 15 min followed by 80°C for 15 min) and Sanger sequencing was performed by the University of Michigan Sequencing Core Facility. Oligonucleotides used for PCR amplification and sequencing are indicated in <xref ref-type="supplementary-material" rid="supp14">Supplementary file 14</xref>.</p></sec><sec id="s4-7"><title>Site-directed mutagenesis</title><p>Thirty-four mutations identified by BSA-Seq and 11 mutations identified by sequencing candidate genes were introduced individually in the genome of the progenitor strain YPW1139 to quantify the effect of these mutations on fluorescence level. ‘Scarless’ genome editing (i.e. without insertion of a selection marker) was achieved using either the delitto perfetto approach from <xref ref-type="bibr" rid="bib73">Stuckey et al., 2011</xref> (for 19 mutations) or CRISPR-Cas9 approaches derived from <xref ref-type="bibr" rid="bib38">Laughery et al., 2015</xref> (for 26 mutations). Compared to delitto perfetto, CRISPR-Cas9 is more efficient and it can be used to introduce mutations in essential genes. However, it requires specific sequences in the vicinity of the target mutation (see below). The technique used for the insertion of each mutation is indicated in <xref ref-type="supplementary-material" rid="supp15">Supplementary file 15</xref>. The sequences of oligonucleotides used for the insertion and the validation of each mutation can be found in <xref ref-type="supplementary-material" rid="supp14">Supplementary file 14</xref>.</p><p>In the delitto perfetto approach, the target site was first replaced by a cassette containing the <italic>Ura3</italic> and <italic>hphMX4</italic> selection markers (pop-in) and then this cassette was swapped with the target mutation (pop-out). The <italic>Ura3-hphMX4</italic> cassette was amplified from pCORE-UH plasmid using two oligonucleotides that contained at their 5’ end 20 nucleotides for PCR priming in pCORE-UH and at their 3’ end 40 nucleotides corresponding to the sequences flanking the target site in the yeast genome (for homologous recombination). The amplicon was transformed into YPW1139 cells using a classic LiAc/polyethylene glycol heat shock protocol (<xref ref-type="bibr" rid="bib19">Gietz and Schiestl, 2007</xref>). Cells were then plated on synthetic complete medium lacking uracil (SC-Ura) and incubated for two days at 30°C. Colonies were replica-plated on YPD + Hygromycin B (300 mg/l) plates. A dozen [Ura+ Hyg+] colonies were streaked on SC-Ura plates to remove residual parental cells and the resulting colonies were patched on YPG plates to counterselect petite cells. Cell patches were then screened by PCR to confirm the proper insertion of <italic>Ura3-hphMX4</italic> at the target site. One positive clone was grown in YPD and stored at −80°C in 15% glycerol. For the pop-out step, a genomic region of ~240 bp centered on the mutation was amplified from the EMS-treated mutant containing the desired mutation. The amplicon was transformed into the strain with <italic>Ura3-hphMX4</italic> inserted at the target site. Cells were plated on a synthetic complete medium containing 0.9 g/l of 5-fluoroorotic acid (SC + 5-FOA) to counterselect cells expressing <italic>Ura3</italic>. After growth, a dozen [Ura-] colonies were streaked on SC + 5-FOA plates and one colony from each streak was patched on a YPG plate. Cell patches were screened by PCR using oligonucleotides that flanked the sequence of the transformed region and amplicons of expected size (~350 bp) were sequenced to confirm the insertion of the desired mutation and the absence of PCR-induced mutations. When possible two independent clones were stored at −80°C in 15% glycerol, but in some cases only one positive clone could be retrieved and stored.</p><p>A ‘one-step’ CRISPR-Cas9 approach was used to insert mutations impairing a NGG or CCN motif in the genome (22 mutations), which corresponds to the protospacer adjacent motif (PAM) targeted by Cas9. First, a DNA fragment containing the 20 bp sequence upstream of the target PAM in the yeast genome was cloned between SwaI and BclI restriction sites in the pML104 plasmid. This DNA fragment was obtained by hybridizing two oligonucleotides designed as described in <xref ref-type="bibr" rid="bib38">Laughery et al., 2015</xref>. The resulting plasmid contained cassettes for expression of Ura3, Cas9 and a guide RNA targeted to the mutation site in yeast cells. In parallel, a repair fragment containing the mutation was obtained either by PCR amplification of a ~240 bp genomic region centered on the mutation in the EMS-treated mutant or by hybridization of two complementary 70-mer oligonucleotides containing the mutation and its flanking genomic sequences. The Cas9/sgRNA plasmid and the repair fragments were transformed together (~150 nmol of plasmid + 20 µmol of repair fragment) into the progenitor strain YPW1139 using LiAc/polyethylene glycol heat shock protocol (<xref ref-type="bibr" rid="bib19">Gietz and Schiestl, 2007</xref>). Cells were then plated on SC-Ura medium and incubated at 30°C for 48 hr. This medium selected cells that both internalized the plasmid and integrated the desired mutation in their genome. Indeed, cells with the Cas9/sgRNA plasmid stop growing as long as their genomic DNA is cleaved by Cas9 but their growth can resume once the PAM sequence is impaired by the mutation, which is integrated into the genome via homologous recombination with the repair fragment (<xref ref-type="bibr" rid="bib38">Laughery et al., 2015</xref>). A dozen [Ura+] colonies were then streaked on SC-Ura plates and one colony from each streak was patched on a YPG plate. Cell patches were screened by PCR using oligonucleotides that flanked the mutation site and amplicons of expected size (~350 bp) were sequenced to confirm the insertion of the desired mutation and the absence of secondary mutations. Then, one or two positive clones were patched on SC + 5-FOA to counterselect the Cas9/sgRNA plasmid, grown in YPD and stored at −80°C in 15% glycerol.</p><p>A ‘two-steps’ CRISPR-Cas9 approach was used to insert mutations located near but outside a PAM sequence (four mutations). Each step was performed as described above for the ‘one-step’ CRISPR-Cas9 approach. In the first step, Cas9 was targeted by the sgRNA to a PAM sequence (the initial PAM) located close to the mutation site (up to 20 bp). The repair fragment contained two synonymous mutations that were not the target mutation: one mutation that impaired the initial PAM and one mutation that introduced a new PAM as close as possible to the target site. This repair fragment was obtained by hybridization of two complementary 90-mer oligonucleotides and transformed into YPW1139. In the second step, Cas9 was targeted to the new PAM. The repair fragment contained three mutations: two mutations that reverted the mutations introduced in the first step and the target mutation. This repair fragment was obtained by hybridization of two complementary 90-mer oligonucleotides and transformed into the strain obtained in the first step. Positive clones were sequenced to confirm the insertion of the target mutation and the absence of other mutations.</p><p>We used CRISPR/Cas9-guided allele replacement to introduce individual mutations in five codons of the <italic>RAP1</italic> coding sequence that encode for amino acids predicted to make direct contact with DNA when RAP1 binds to DNA (<xref ref-type="bibr" rid="bib34">Konig et al., 1996</xref>). For each codon, we tried to insert one synonymous mutation, one nonsynonymous mutation predicted to have a weak impact on RAP1 protein structure and one nonsynonymous mutation predicted to have a strong impact on RAP1 protein structure based on amino acid exchangeability scores from <xref ref-type="bibr" rid="bib84">Yampolsky and Stoltzfus, 2005</xref> (see <xref ref-type="supplementary-material" rid="supp8">Supplementary file 8</xref> for the list of mutations). Each mutation was introduced in the genome of strain YPW2706. This strain is derived from YPW1139 and contains two identical sgRNA target sites upstream and downstream of the <italic>RAP1</italic> gene (see below for details on YPW2706 construction). Therefore, we could use a single Cas9/sgRNA plasmid to excise the entire <italic>RAP1</italic> gene in YPW2706 by targeting Cas9 to both ends of the gene. We used gene SOEing (Splicing by Overlap Extension) to generate repair fragments corresponding to the <italic>RAP1</italic> gene (promoter and coding sequence) with each target mutation. First, a left fragment of <italic>RAP1</italic> was amplified from YPW1139 genomic DNA using a forward 20-mer oligonucleotide priming upstream of the RAP1 promoter and a reverse 60-mer oligonucleotide containing the target mutation and the surrounding <italic>RAP1</italic> sequence. In parallel, a right fragment of RAP1 overlapping with the right fragment was amplified from YPW1139 genomic DNA using a forward 60-mer oligonucleotide complementary to the reverse oligonucleotide used to amplify the left fragment and a reverse 20-mer oligonucleotide priming in <italic>RAP1</italic> 5’UTR sequence. Then, equimolar amounts of the left and right fragments were mixed in a PCR reaction and 25 cycles of PCR were performed to fuse both fragments. Finally, the resulting product was further amplified using two 90-mer oligonucleotides with homology to the sequence upstream of <italic>RAP1</italic> promoter and to the <italic>RAP1</italic> 5’UTR but without the sgRNA target sequences. Consequently, transformation of the repair fragment together with the Cas9/sgRNA plasmid in YPW2706 cells was expected to replace the wild type allele of <italic>RAP1</italic> by an allele containing the target mutation in <italic>RAP1</italic> coding sequence and without the two flanking sgRNA target sites. For each of the 15 target mutations, we sequenced the <italic>RAP1</italic> promoter and coding sequence in 10 independent clones obtained after transformation. All synonymous mutations were retrieved in several clones, while several of the nonsynonymous mutations were not found in any clone, suggesting they were lethal (<xref ref-type="supplementary-material" rid="supp8">Supplementary file 8</xref>).</p></sec><sec id="s4-8"><title><italic>RAP1</italic> and <italic>GCR1</italic> mutagenesis using error-prone PCR</title><p>We used a mutagenic PCR approach to efficiently generate hundreds of mutants with random mutations in the <italic>RAP1</italic> gene (promoter and coding sequence) or in the second exon of <italic>GCR1</italic> (representing 99.7% of <italic>GCR1</italic> coding sequence). DNA fragments obtained from the mutagenic PCR were introduced in the yeast genome using CRISPR/Cas9-guided allele replacement as described above. The sequences of all oligonucleotides used for <italic>RAP1</italic> and <italic>GCR1</italic> mutagenesis can be found in <xref ref-type="supplementary-material" rid="supp14">Supplementary file 14</xref>.</p><p>First, we constructed two yeast strains for which the <italic>RAP1</italic> gene (strain YPW2706) or the second exon of <italic>GCR1</italic> (strain YPW3082) were flanked by identical sgRNA target sites and PAM sequences. To generate strain YPW2706, we first identified a sgRNA target site located downstream of the <italic>RAP1</italic> coding sequence (41 bp after the stop codon in the 5’UTR) in the S288c genome. Then, we inserted the 23 bp sequence corresponding to this sgRNA target site and PAM upstream of the <italic>RAP1</italic> promoter (immediately after <italic>PPN2</italic> stop codon) in strain YPW1139 using the delitto perfetto approach (as described above). To generate strain YPW3082, we first identified a sgRNA target site located at the end of the <italic>GCR1</italic> intron (22 bp upstream of exon 2) in the S288c genome. Then, we inserted the 23 bp sequence corresponding to this sgRNA target site and PAM immediately after the <italic>GCR1</italic> stop codon in strain YPW1139 using the delitto perfetto approach (as described above).</p><p>Second, we constructed plasmid pPW437 by cloning the 20mer guide sequence directed to <italic>RAP1</italic> in pML104 as described in <xref ref-type="bibr" rid="bib38">Laughery et al., 2015</xref> and we constructed plasmid pPW438 by cloning the 20mer guide sequence directed to <italic>GCR1</italic> in pML104 as described in <xref ref-type="bibr" rid="bib38">Laughery et al., 2015</xref>. These two sgRNA/Cas9 plasmids can be used, respectively, to excise the <italic>RAP1</italic> gene or <italic>GCR1</italic> exon two from the genomes of YPW2706 and YPW3082.</p><p>Third, we generated repair fragments with random mutations in <italic>RAP1</italic> or <italic>GCR1</italic> genes using error-prone PCR. We first amplified each gene from 2 ng of YPW1139 genomic DNA using a high-fidelity polymerase (KAPA HiFi DNA polymerase) and 30 cycles of PCR. PCR products were purified with the Wizard SV Gel and PCR Clean-Up System (Promega) and quantified with a Qubit 2.0 Fluorometer (Thermo Fisher Scientific) using dsDNA broad range assays. Two nanograms of purified PCR products were used as template for a first round of mutagenic PCR and mixed with 25 µl of DreamTaq Master Mix 2x (ThermoFisher Scientific), 2.5 µl of forward and reverse primers at 10 µM, 5 µl of 1 mM dATP and 5 µl of 1 mM dTTP in a final volume of 50 µl. The imbalance of dNTP concentrations (0.3 µM dATP, 0.2 µM dCTP, 0.2 µM dGTP and 0.3 µM dTTP) was done to bias the mutagenesis toward misincorporation of dATP and dTTP. For <italic>RAP1</italic> mutagenesis, the forward oligonucleotide primed upstream of the <italic>RAP1</italic> promoter (in <italic>PPN2</italic> coding sequence) and the reverse oligonucleotide primed in the <italic>RAP1</italic> terminator and contained a mutation in the PAM adjacent to the sgRNA target site. For <italic>GCR1</italic> mutagenesis, the forward oligonucleotide primed at the end of the <italic>GCR1</italic> intron and contained a mutation in the PAM adjacent to the sgRNA target site and the reverse primer primed in the <italic>GCR1</italic> terminator. The PCR program was 95°C for 3 min followed by 32 cycles with 95°C for 30 s, 52°C for 30 s, 72°C for 2 min and a final extension at 72°C for 5 min. For <italic>RAP1</italic> mutagenesis, the product of the first mutagenic PCR was diluted by a factor of 33 and used as template for a second round of mutagenic PCR (1.5 µl of product in a 50 µl reaction) similar to the first round but with only 10 cycles of amplification. For <italic>GCR1</italic> mutagenesis, the product of the first mutagenic PCR was diluted by a factor of 23 and used as template for a second round of mutagenic PCR (2.2 µl of product in a 50 µl reaction) with 35 cycles of amplification. Using this protocol, we expected to obtain on average 1.6 mutations per fragment for <italic>RAP1</italic> mutagenesis and 1.8 mutations per fragment for <italic>GCR1</italic> mutagenesis (see below for calculations of these estimates).</p><p>pPW437 was transformed with <italic>RAP1</italic> repair fragments into YPW2706 and pPW438 was transformed with <italic>GCR1</italic> repair fragments into YPW3082 as described above for CRISPR/Cas9 site directed mutagenesis. To select cells that replaced the wild type alleles with alleles containing random mutations, transformed cells were plated on SC-Ura and incubated at 30°C for 48 hr. To confirm the success of each mutagenesis and to estimate actual mutation rates, we then sequenced the <italic>RAP1</italic> genes in 27 random colonies from the <italic>RAP1</italic> mutagenesis and we sequenced the second exon of <italic>GCR1</italic> in 18 random colonies from the <italic>GCR1</italic> mutagenesis. Next, 500 colonies from <italic>RAP1</italic> mutagenesis and 300 colonies from <italic>GCR1</italic> mutagenesis were streaked onto SC-Ura plates. After growth, one colony from each streak was patched on YPG and grown four days at 30°C. Then, patches were replica-plated with velvets onto SC + 5-FOA to eliminate sgRNA/Cas9 plasmids. Finally, 488 clones from <italic>RAP1</italic> mutagenesis and 355 clones from <italic>GCR1</italic> mutagenesis were arrayed in 96-well plates containing 0.5 ml of YPD (same plate design as used for the flow cytometry assays) and grown overnight at 30°C. 0.2 ml of cell culture from each well was then mixed with 46 µl of 80% glycerol in 96-well plates and stored at −80°C. The fluorescence of these strains was quantified by flow cytometry as described above to assess the impact of <italic>RAP1</italic> and <italic>GCR1</italic> mutations on <italic>P<sub>TDH3</sub>-YFP</italic> expression (expression data for each mutant can be found in <xref ref-type="supplementary-material" rid="supp16">Supplementary file 16</xref>).</p><p>In our mutagenesis approach, we introduced a mutation that impaired the target PAM sequence in all <italic>RAP1</italic> and <italic>GCR1</italic> mutants. To determine the effect of this mutation alone, we generated strains YPW2701 and YPW2732 that carried the PAM mutation in the <italic>RAP1</italic> terminator or in the <italic>GCR1</italic> intron, respectively, without any other mutation in <italic>RAP1</italic> or <italic>GCR1</italic>. The fluorescence level of these two strains was not significantly different from the fluorescence level of the progenitor strain YPW1139 in flow cytometry assays.</p></sec><sec id="s4-9"><title>Estimation of <italic>RAP1</italic> and <italic>GCR1</italic> mutation rates</title><p>The expected number of mutations per PCR amplicon (<inline-formula><mml:math id="inf1"><mml:msub><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mi>u</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>) depends on the error rate of the Taq polymerase (μ), on the number of DNA duplications (<inline-formula><mml:math id="inf2"><mml:mi>D</mml:mi></mml:math></inline-formula>) and on the length of the amplicon (<inline-formula><mml:math id="inf3"><mml:mi>L</mml:mi></mml:math></inline-formula>):<inline-formula><mml:math id="inf4"><mml:msub><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mi>u</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mi>μ</mml:mi><mml:mo>⋅</mml:mo><mml:mi>D</mml:mi><mml:mo>⋅</mml:mo><mml:mi>L</mml:mi></mml:math></inline-formula>. The published error rate for a classic polymerase similar to DreamTaq is ~3 x 10<sup>−5</sup> errors per nucleotide per duplication (<xref ref-type="bibr" rid="bib49">McInerney et al., 2014</xref>). Amplicon length was 3057 bp for <italic>RAP1</italic> mutagenesis and 2520 pb for <italic>GCR1</italic> mutagenesis. The number of duplications of PCR templates was calculated from the amounts of double stranded DNA quantified using Qubit 2.0 dsDNA assays before (<inline-formula><mml:math id="inf5"><mml:mi>I</mml:mi></mml:math></inline-formula>) and after (<inline-formula><mml:math id="inf6"><mml:mi>O</mml:mi></mml:math></inline-formula>) each mutagenic PCR reaction as follows: <inline-formula><mml:math id="inf7"><mml:mi>D</mml:mi><mml:mo>=</mml:mo><mml:mi>l</mml:mi><mml:mi>n</mml:mi><mml:mfenced separators="|"><mml:mrow><mml:mfrac><mml:mrow><mml:mi>O</mml:mi></mml:mrow><mml:mrow><mml:mi>I</mml:mi></mml:mrow></mml:mfrac></mml:mrow></mml:mfenced><mml:mo>÷</mml:mo><mml:mi>l</mml:mi><mml:mi>n</mml:mi><mml:mi/><mml:mn>2</mml:mn></mml:math></inline-formula>. For the first round of <italic>RAP1</italic> mutagenesis,<inline-formula><mml:math id="inf8"><mml:mi>D</mml:mi><mml:mo>=</mml:mo><mml:mi>l</mml:mi><mml:mi>n</mml:mi><mml:mfenced separators="|"><mml:mrow><mml:mfrac><mml:mrow><mml:mn>6550</mml:mn></mml:mrow><mml:mrow><mml:mn>1.93</mml:mn></mml:mrow></mml:mfrac></mml:mrow></mml:mfenced><mml:mo>÷</mml:mo><mml:mi>l</mml:mi><mml:mi>n</mml:mi><mml:mi/><mml:mn>2</mml:mn><mml:mi/><mml:mo>=</mml:mo><mml:mn>11.7</mml:mn></mml:math></inline-formula>. For the second round of <italic>RAP1</italic> mutagenesis, <inline-formula><mml:math id="inf9"><mml:mi>D</mml:mi><mml:mo>=</mml:mo><mml:mi>l</mml:mi><mml:mi>n</mml:mi><mml:mfenced separators="|"><mml:mrow><mml:mfrac><mml:mrow><mml:mn>3000</mml:mn></mml:mrow><mml:mrow><mml:mn>68.1</mml:mn></mml:mrow></mml:mfrac></mml:mrow></mml:mfenced><mml:mo>÷</mml:mo><mml:mi>l</mml:mi><mml:mi>n</mml:mi><mml:mi/><mml:mn>2</mml:mn><mml:mi/><mml:mo>=</mml:mo><mml:mn>5.5</mml:mn></mml:math></inline-formula>. Therefore, the total number of duplications was 17.2 and the expected number of mutations per amplicon <inline-formula><mml:math id="inf10"><mml:msub><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mi>u</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> was 1.6 on average. For the first round of <italic>GCR1</italic> mutagenesis, <inline-formula><mml:math id="inf11"><mml:mi>D</mml:mi><mml:mo>=</mml:mo><mml:mi>l</mml:mi><mml:mi>n</mml:mi><mml:mfenced separators="|"><mml:mrow><mml:mfrac><mml:mrow><mml:mn>6870</mml:mn></mml:mrow><mml:mrow><mml:mn>2.15</mml:mn></mml:mrow></mml:mfrac></mml:mrow></mml:mfenced><mml:mo>÷</mml:mo><mml:mi>l</mml:mi><mml:mi>n</mml:mi><mml:mi/><mml:mn>2</mml:mn><mml:mi/><mml:mo>=</mml:mo><mml:mn>11.6</mml:mn></mml:math></inline-formula>. For the second round of <italic>GCR1</italic> mutagenesis, <inline-formula><mml:math id="inf12"><mml:mi>D</mml:mi><mml:mo>=</mml:mo><mml:mi>l</mml:mi><mml:mi>n</mml:mi><mml:mfenced separators="|"><mml:mrow><mml:mfrac><mml:mrow><mml:mn>6535</mml:mn></mml:mrow><mml:mrow><mml:mn>1.65</mml:mn></mml:mrow></mml:mfrac></mml:mrow></mml:mfenced><mml:mo>÷</mml:mo><mml:mi>l</mml:mi><mml:mi>n</mml:mi><mml:mi/><mml:mn>2</mml:mn><mml:mi/><mml:mo>=</mml:mo><mml:mn>12.0</mml:mn></mml:math></inline-formula>. Therefore, the total number of duplications was 23.6 and the expected number of mutations per amplicon <inline-formula><mml:math id="inf13"><mml:msub><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mi>u</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> was 1.8 on average.</p></sec><sec id="s4-10"><title>Effects of mutations in purine biosynthesis genes on expression from different promoters</title><p>We compared the individual effects of three mutations in the purine biosynthesis pathway (<italic>ADE2-C1477a</italic>, <italic>ADE5-G1715a</italic> and <italic>ADE6-G3327a</italic>) on YFP expression driven by four different yeast promoters (<italic>P<sub>TDH3</sub></italic>, <italic>P<sub>RNR1</sub></italic>, <italic>P<sub>STM1</sub></italic>, and <italic>P<sub>GPD1</sub></italic>). Each mutation was introduced individually in the genomes of four parental strains described in <xref ref-type="bibr" rid="bib23">Hodgins-Davis et al., 2019</xref> carrying either <italic>P<sub>TDH3</sub>-YFP</italic> (YPW1139), <italic>P<sub>RNR1</sub>-YFP</italic> (YPW3758), <italic>P<sub>STM1</sub>-YFP</italic> (YPW3764), or <italic>P<sub>GPD1</sub>-YFP</italic> (YPW3757) reporter gene at the <italic>ho</italic> locus. Site-directed mutagenesis was performed as described in the corresponding section (see above). The fluorescence of the four parental strains, of a non-fluorescent strain (YPW978) and of the 12 mutant strains (four reporter genes x three mutations) was quantified using a Sony MA-900 flow cytometer (the BD Accuri C6 instrument used for other fluorescence assays was not available due to Covid-19 shutdown) in three replicate experiments performed on different days. For each experiment, all strains were grown in parallel in culture tubes containing 5 ml of YPD and incubated at 30°C for 16 hr. Each sample was diluted to 1–2 x 10<sup>7</sup> cells/mL in PBS prior to measurement. At least 5 x 10<sup>4</sup> events were recorded for each sample using a 488 nm laser for YFP excitation and a 525/50 optical filter for the acquisition of fluorescence. At least 5 x 10<sup>4</sup> events were recorded for each sample. Flow cytometry data were then processed in R using functions from the <italic>FlowCore</italic> package and custom scripts available in <xref ref-type="supplementary-material" rid="scode1">Source code 1</xref>. After log-transformation of flow data, events considered to correspond to single cells were selected on the basis of their forward scatter height and width (FSC-H and FSC-W). Fluorescence values of single cells were then normalized to account for differences in cell size. Finally, the median fluorescence among cells was computed for each sample and averaged across replicates of each genotype.</p></sec><sec id="s4-11"><title>Statistical comparisons of <italic>trans</italic>-regulatory and nonregulatory mutations</title><p>We established a set of 69 <italic>trans</italic>-regulatory mutations that included 52 mutations with a p-value below 0.01 in the <italic>G</italic>-tests comparing the frequencies of mutant and reference alleles in low and high fluorescence bulks (see above) as well as 17 mutations identified by Sanger sequencing in the coding sequence of purine biosynthesis genes. In parallel, we established a set of 1766 nonregulatory mutations regarding <italic>P<sub>TDH3</sub>-YFP</italic> expression that included mutations with a p-value above 0.01 in the <italic>G</italic>-tests comparing the frequencies of mutant and reference alleles in low and high fluorescence bulks (see above) and mutations that did not affect <italic>P<sub>TDH3</sub>-YFP</italic> expression in single-site mutants. We performed statistical analysis to compare properties of <italic>trans</italic>-regulatory and nonregulatory mutations using RStudio v1.2.5019 (R scripts are in <xref ref-type="supplementary-material" rid="scode2">Source code 2</xref>). We used <italic>G</italic>-tests (<italic>likelihood.ratio</italic> function in <italic>Deducer</italic> package) to compare the following properties between <italic>trans</italic>-regulatory and nonregulatory mutations: (i) the frequency of G:C to A:T transitions, (ii) the frequency of indels, (iii) the frequency of aneuploidies, (iv) the distribution of mutations among chromosomes, (v) the frequency of mutations in coding, intronic and intergenic sequences, (vi) the frequency of synonymous, nonsynonymous and nonsense changes among coding mutations, (vii) the frequency of coding mutations in transcription factors, (viii) the frequency of coding mutations in the predicted <italic>TDH3</italic> regulatory network (see below), (ix) the proportion of mutations in eQTL regions (see below). We used resampling tests to compare the frequencies of different amino acid changes caused by <italic>trans</italic>-regulatory and nonregulatory mutations in coding sequences. We computed for each possible amino acid change the observed absolute difference between (i) the proportion of coding <italic>trans-</italic>regulatory mutations causing the amino acid change and (ii) the proportion of nonregulatory mutations causing the amino acid change. Then, we computed similar absolute differences for 10,000 randomly permuted sets of <italic>trans</italic>-regulatory and nonregulatory mutations. The p-value for each amino acid change was calculated as the proportion of resampled absolute differences greater or equal to the observed absolute difference.</p></sec><sec id="s4-12"><title><italic>TDH3</italic> regulatory network</title><p>The network of potential <italic>TDH3</italic> regulators shown on <xref ref-type="fig" rid="fig4">Figure 4</xref> was established using data available in July 2019 on the YEASTRACT (<ext-link ext-link-type="uri" xlink:href="http://www.yeastract.com/">http://www.yeastract.com/</ext-link>) repository of regulatory associations between transcription factors and target genes in <italic>Saccharomyces cerevisiae</italic> (<xref ref-type="bibr" rid="bib76">Teixeira et al., 2018</xref>). We used the tool ‘Regulation Matrix’ to obtain three matrices in which rows corresponded to the 220 transcription factor genes in YEASTRACT and columns corresponded to the 6886 yeast target genes included in the database. In the first matrix obtained using the option ‘Only DNA binding evidence’, an element had a value of 1 if the transcription factor at the corresponding row was reported in the literature to bind to the promoter of the target gene at the corresponding column and a value of 0 otherwise. The two other matrices were obtained using the option ‘Only Expression evidence’ with either ‘TF acting as activator’ or ‘TF acting as inhibitor’. An element had a value of 1 only in the ‘TF acting as activator’ matrix if perturbation of the transcription factor at the corresponding row was reported to increase expression of the target gene at the corresponding column. An element had a value of 1 only in the ‘TF acting as inhibitor’ matrix if perturbation of the transcription factor at the corresponding row was reported to decrease expression of the target gene at the corresponding column. An element had a value of 1 in both matrices if perturbation of the transcription factor at the corresponding row was reported to affect expression of the target gene at the corresponding column in an undetermined direction. Finally, an element had a value of 0 in both matrices if perturbation of the transcription factor at the corresponding row was not reported to alter expression of the target gene at the corresponding column in the literature. We then used a custom R script (<xref ref-type="supplementary-material" rid="scode2">Source code 2</xref>) to generate a smaller matrix that only contained first level and second level regulators of <italic>TDH3</italic> and <italic>TDH3</italic> itself. A transcription factor was considered to be a first level regulator of <italic>TDH3</italic> if a regulatory association with <italic>TDH3</italic> was supported both by DNA binding evidence and expression evidence. A transcription factor was considered to be a second level regulator of <italic>TDH3</italic> if a regulatory association with a first level regulator of <italic>TDH3</italic> was supported both by DNA binding evidence and expression evidence. The network shown on <xref ref-type="fig" rid="fig4">Figure 4</xref> was drawn using Adobe Illustrator based on regulatory interactions included in the matrix of <italic>TDH3</italic> regulators (in <xref ref-type="supplementary-material" rid="supp12">Supplementary file 12</xref>). To determine whether mutations in the <italic>TDH3</italic> regulatory network constituted a significant mutational source of regulatory variation affecting <italic>P<sub>TDH3</sub></italic> activity, we compared the proportions of <italic>trans</italic>-regulatory and non-regulatory mutations that were located in a <italic>TDH3</italic> regulator gene (first or second level) using a <italic>G</italic>-test (<italic>likelihood.ratio</italic> function in R package <italic>Deducer</italic>).</p></sec><sec id="s4-13"><title>Competitive fitness assays</title><p>We performed competitive growth assays to quantify the fitness of 62 strains with random mutations in the second exon of <italic>GCR1</italic>. These 62 strains corresponded to all <italic>GCR1</italic> mutants that showed a significant decrease of <italic>P<sub>TDH3</sub>-YFP</italic> expression as quantified by flow cytometry as well as <italic>GCR1</italic> mutants for which <italic>GCR1</italic> exon 2 was sequenced and the location of mutations was known. The 62 strains were thawed on YPG plates as well as reference strains YPW1139 and YPW2732 and strain YPW1182 that expressed a GFP (Green Fluorescent Protein) reporter instead of YFP. After 3 days of incubation at 30°C, strains were arrayed in four replicate 96-well plates containing 0.5 ml of YPG per well. In parallel, the [GFP+] strain YPW1182 was also arrayed in four replicate 96-well plates. The eight plates were incubated on a wheel at 30°C for 32 hours. We then measured the optical density at 620 nm of all samples using a Sunrise plate reader (Tecan) and calculated the average cell density for each plate. Samples were then transferred to 1.2 ml of YPD in 96-well plates to reach an average cell density of 10<sup>6</sup> cells/ml for each plate. 21.25 µl of samples from plates containing [YFP+] strains were mixed with 3.75 µl of [GFP+] samples in four 96-well plates containing 0.45 ml of YPD per well. The reason why [YFP+] and [GFP+] strains were mixed to a 17:3 ratio is because we anticipated that some of the <italic>GCR1</italic> mutants may grow slower than the [GFP+] competitor in YPD. Samples were then grown on a wheel at 30°C for 10 hr and the optical density was measured again after growth to estimate the average number of generations for each plate. The ratio of [YFP+] and [GFP+] cells in each sample was quantified by flow cytometry before and after the 10 hr of growth. Samples were analyzed on a BD Accuri C6 flow cytometer with a 488 nm laser used for excitation and two different optical filters (510/10 and 585/40) used to separate YFP and GFP signals. FCS data were analyzed with custom R scripts using <italic>flowCore</italic> and <italic>flowClust</italic> packages (<xref ref-type="supplementary-material" rid="scode1">Source code 1</xref>) as described in <xref ref-type="bibr" rid="bib11">Duveau et al., 2018</xref>. First, we filtered out artifactual events with extreme values of forward scatter or fluorescence intensity. Then, for each sample we identified two clusters of events corresponding to [YFP+] and [GFP+] cells using a principal component analysis on the logarithms of FL1.H and FL2.H (height of the fluorescence signal captured through the 510/10 and 585/40 filters, respectively). Indeed, [YFP+] cells tend to have lower FL1.H value and higher FL2.H value than [GFP+] cells and these two parameters are positively correlated. The competitive fitness of [YFP+] cells relative to [GFP+] cells was calculated as the exponential of the slope of the linear regression of <inline-formula><mml:math id="inf14"><mml:msub><mml:mrow><mml:mi>l</mml:mi><mml:mi>o</mml:mi><mml:mi>g</mml:mi></mml:mrow><mml:mrow><mml:mi>e</mml:mi></mml:mrow></mml:msub><mml:mfenced separators="|"><mml:mrow><mml:mfrac><mml:mrow><mml:mi>Y</mml:mi><mml:mi>F</mml:mi><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mi>G</mml:mi><mml:mi>F</mml:mi><mml:mi>P</mml:mi></mml:mrow></mml:mfrac></mml:mrow></mml:mfenced></mml:math></inline-formula> on the number of generations of growth (where <inline-formula><mml:math id="inf15"><mml:mi>Y</mml:mi><mml:mi>F</mml:mi><mml:mi>P</mml:mi></mml:math></inline-formula> corresponds to the number of [YFP+] cells and <inline-formula><mml:math id="inf16"><mml:mi>G</mml:mi><mml:mi>F</mml:mi><mml:mi>P</mml:mi></mml:math></inline-formula> corresponds to the number of [GFP+] cells). We then divided the fitness of each sample by the mean fitness among all replicates of the reference strain YPW1139 to obtain a fitness value relative to YPW1139. The fitness of each strain was calculated as the mean relative fitness among the four replicate populations for that strain. These fitness data can be found in <xref ref-type="supplementary-material" rid="supp16">Supplementary file 16</xref>.</p></sec><sec id="s4-14"><title>Gene ontology (GO) analysis</title><p>GO term analyses were performed on <ext-link ext-link-type="uri" xlink:href="http://www.pantherdb.org/">http://www.pantherdb.org/</ext-link> website in June 2020 (<xref ref-type="bibr" rid="bib55">Mi et al., 2019</xref>). In ‘Gene List Analysis’, we used ‘Statistical overrepresentation test’ on a query list corresponding to the 42 genes affected by <italic>trans</italic>-regulatory coding mutations. GO enrichment was determined based on a reference list of the 1251 genes affected by non-regulatory coding mutations using Fisher’s exact tests. Four separate analyses were performed for GO biological processes, GO molecular functions, GO cellular components and PANTHER pathways. GO terms that are significantly enriched in the list of <italic>trans</italic>-regulatory mutations (mutations associated with fluorescence level) relative to non-regulatory mutations (mutations not associated with fluorescence level) at p &lt; 0.05 are listed in <xref ref-type="supplementary-material" rid="supp9">Supplementary file 9</xref>.</p></sec><sec id="s4-15"><title>Enrichment of mutations in eQTL regions</title><p>Genomic regions containing expression quantitative trait loci (eQTL) associated with <italic>P<sub>TDH3</sub>-YFP</italic> expression variation in three different crosses (BYxYPS1000, BYxSK1 and BYxM22) were obtained from <xref ref-type="supplementary-material" rid="supp11">Supplementary file 11</xref> in <xref ref-type="bibr" rid="bib54">Metzger and Wittkopp, 2019</xref>. A custom R script was used to determine the number of <italic>trans</italic>-regulatory and non-regulatory mutations located inside and outside these eQTL intervals (<xref ref-type="supplementary-material" rid="scode2">Source code 2</xref>). <italic>G</italic>-tests were performed to determine whether the proportion of <italic>trans</italic>-regulatory mutations in eQTL intervals was statistically different from the proportion of non-regulatory mutations in the same eQTL intervals.</p></sec><sec id="s4-16"><title>Data archiving</title><p>De-multiplexed sequencing data are available in FASTQ format from NCBI Sequence Read Archive (<ext-link ext-link-type="uri" xlink:href="https://www.ncbi.nlm.nih.gov/sra">https://www.ncbi.nlm.nih.gov/sra</ext-link>) under BioProject number PRJNA706682. Flow cytometry data (FCS files) are available on the Flow Repository (<ext-link ext-link-type="uri" xlink:href="https://flowrepository.org/">https://flowrepository.org/</ext-link>) under the following experiments ID: FR-FCM-Z3WV for the secondary screen of fluorescence in EMS mutants shown in <xref ref-type="fig" rid="fig1">Figure 1E</xref>, FR-FCM-Z3JY for the quantifications of fluorescence in single site mutants and in the corresponding EMS mutants (<xref ref-type="fig" rid="fig2">Figure 2E–G</xref>), FR-FCM-Z3J2 for the quantifications of fluorescence in <italic>RAP1</italic> mutant strains (<xref ref-type="fig" rid="fig5">Figure 5E</xref>), FR-FCM-Z3J3 for the quantifications of fluorescence in <italic>GCR1</italic> mutant strains (<xref ref-type="fig" rid="fig5">Figure 5F–G</xref>) and FR-FCM-Z3J5 for the quantifications of fitness in the same <italic>GCR1</italic> mutants strains (<xref ref-type="fig" rid="fig5">Figure 5G</xref>).</p></sec></sec></body><back><ack id="ack"><title>Acknowledgements</title><p>We thank Gaël Yvert and Mark Hill for helpful comments on the manuscript, the University of Michigan sequencing core and University of Michigan flow cytometry core for research support, and the National Institutes of Health (R01GM108826 and R35GM118073 to PJW), European Molecular Biology Organization (EMBO ALTF 1114–2012 to FD), National Science Foundation (MCB-1929737 to PJW), NIH Genetics Training grant (T32GM007544 to PVZ), NIH Genome Sciences Training Grant (T32HG000040 to BPHM and MAS), and the Michigan Life Sciences Fellow program (MAS) for funding.</p></ack><sec id="s5" sec-type="additional-information"><title>Additional information</title><fn-group content-type="competing-interest"><title>Competing interests</title><fn fn-type="COI-statement" id="conf2"><p>Senior editor, <italic>eLife</italic></p></fn><fn fn-type="COI-statement" id="conf1"><p>No competing interests declared</p></fn></fn-group><fn-group content-type="author-contribution"><title>Author contributions</title><fn fn-type="con" id="con1"><p>Conceptualization, Data curation, Formal analysis, Supervision, Validation, Investigation, Visualization, Methodology, Writing - original draft, Writing - review and editing</p></fn><fn fn-type="con" id="con2"><p>Formal analysis, Validation, Investigation, Methodology, Writing - review and editing</p></fn><fn fn-type="con" id="con3"><p>Conceptualization, Investigation, Methodology, Writing - review and editing</p></fn><fn fn-type="con" id="con4"><p>Validation, Investigation</p></fn><fn fn-type="con" id="con5"><p>Validation, Investigation</p></fn><fn fn-type="con" id="con6"><p>Validation, Investigation</p></fn><fn fn-type="con" id="con7"><p>Formal analysis, Validation, Investigation, Methodology</p></fn><fn fn-type="con" id="con8"><p>Validation, Investigation</p></fn><fn fn-type="con" id="con9"><p>Conceptualization, Supervision, Funding acquisition, Methodology, Writing - original draft, Project administration, Writing - review and editing</p></fn></fn-group></sec><sec id="s6" sec-type="supplementary-material"><title>Additional files</title><supplementary-material id="scode1"><label>Source code 1.</label><caption><title>R scripts used for the analysis of flow cytometry data.</title></caption><media mime-subtype="zip" mimetype="application" xlink:href="elife-67806-code1-v1.txt.zip"/></supplementary-material><supplementary-material id="scode2"><label>Source code 2.</label><caption><title>R scripts used for the analysis of BSA-Seq data and for comparing the properties of <italic>trans</italic>-regulatory and non-regulatory mutations.</title></caption><media mime-subtype="zip" mimetype="application" xlink:href="elife-67806-code2-v1.txt.zip"/></supplementary-material><supplementary-material id="scode3"><label>Source code 3.</label><caption><title>R script used to annotate variants identified in BSA-Seq data.</title></caption><media mime-subtype="zip" mimetype="application" xlink:href="elife-67806-code3-v1.txt.zip"/></supplementary-material><supplementary-material id="scode4"><label>Source code 4.</label><caption><title>PBS script used to process FASTQ files.</title></caption><media mime-subtype="zip" mimetype="application" xlink:href="elife-67806-code4-v1.txt.zip"/></supplementary-material><supplementary-material id="sdata1"><label>Source data 1.</label><caption><title>Compressed folder including 34.</title><p>Source Data files in txt format that contain quantitative data displayed on <xref ref-type="fig" rid="fig1">Figure 1B–E</xref>, <xref ref-type="fig" rid="fig2">Figure 2E–G</xref>, <xref ref-type="fig" rid="fig3">Figure 3A,B,E</xref>, <xref ref-type="fig" rid="fig5">Figure 5C–G</xref>, <xref ref-type="fig" rid="fig6">Figure 6</xref>, <xref ref-type="fig" rid="fig7">Figure 7</xref>, <xref ref-type="fig" rid="fig2s1">Figure 2—figure supplements 1</xref>–<xref ref-type="fig" rid="fig2s5">5</xref>, <xref ref-type="fig" rid="fig3s1">Figure 3—figure supplements 1</xref>–<xref ref-type="fig" rid="fig3s4">4</xref> and <xref ref-type="fig" rid="fig7s1">Figure 7—figure supplement 1</xref>.</p></caption><media mime-subtype="zip" mimetype="application" xlink:href="elife-67806-data1-v1.zip"/></supplementary-material><supplementary-material id="supp1"><label>Supplementary file 1.</label><caption><title>Sequencing depth in BSA-seq data.</title></caption><media mime-subtype="excel" mimetype="application" xlink:href="elife-67806-supp1-v1.xls"/></supplementary-material><supplementary-material id="supp2"><label>Supplementary file 2.</label><caption><title>List of all mutations identified by BSA-Seq or Sanger sequencing in this study.</title></caption><media mime-subtype="excel" mimetype="application" xlink:href="elife-67806-supp2-v1.xls"/></supplementary-material><supplementary-material id="supp3"><label>Supplementary file 3.</label><caption><title>Statistical associations between aneuploidies and fluorescence level.</title></caption><media mime-subtype="docx" mimetype="application" xlink:href="elife-67806-supp3-v1.docx"/></supplementary-material><supplementary-material id="supp4"><label>Supplementary file 4.</label><caption><title>Linked mutations associated with fluorescence level in BSA-Seq experiments.</title></caption><media mime-subtype="excel" mimetype="application" xlink:href="elife-67806-supp4-v1.xls"/></supplementary-material><supplementary-material id="supp5"><label>Supplementary file 5.</label><caption><title>Mutations identified by Sanger sequencing of candidate genes.</title></caption><media mime-subtype="excel" mimetype="application" xlink:href="elife-67806-supp5-v1.xls"/></supplementary-material><supplementary-material id="supp6"><label>Supplementary file 6.</label><caption><title>Mutations tested in single-site mutants.</title></caption><media mime-subtype="excel" mimetype="application" xlink:href="elife-67806-supp6-v1.xls"/></supplementary-material><supplementary-material id="supp7"><label>Supplementary file 7.</label><caption><title>Mutations associated with fluorescence level in BSA-Seq experiments.</title></caption><media mime-subtype="excel" mimetype="application" xlink:href="elife-67806-supp7-v1.xls"/></supplementary-material><supplementary-material id="supp8"><label>Supplementary file 8.</label><caption><title>Targeted mutagenesis of RAP1 residues making direct contact with DNA.</title></caption><media mime-subtype="excel" mimetype="application" xlink:href="elife-67806-supp8-v1.xls"/></supplementary-material><supplementary-material id="supp9"><label>Supplementary file 9.</label><caption><title>List of GO terms overrepresented in genes hit by causative mutations relative to genes hit by neutral mutations.</title></caption><media mime-subtype="excel" mimetype="application" xlink:href="elife-67806-supp9-v1.xls"/></supplementary-material><supplementary-material id="supp10"><label>Supplementary file 10.</label><caption><title>Mutations located in the coding sequence of glucose signaling genes.</title></caption><media mime-subtype="excel" mimetype="application" xlink:href="elife-67806-supp10-v1.xls"/></supplementary-material><supplementary-material id="supp11"><label>Supplementary file 11.</label><caption><title><italic>Trans</italic>-regulatory effects of mutations in purine biosynthesis genes or iron homeostasis genes.</title></caption><media mime-subtype="excel" mimetype="application" xlink:href="elife-67806-supp11-v1.xls"/></supplementary-material><supplementary-material id="supp12"><label>Supplementary file 12.</label><caption><title>Files used as inputs for analyses performed with the PBS script (<xref ref-type="supplementary-material" rid="scode4">Source code 4</xref>) and R scripts (<xref ref-type="supplementary-material" rid="scode1">Source code 1</xref>–<xref ref-type="supplementary-material" rid="scode3">3</xref>).</title></caption><media mime-subtype="octet-stream" mimetype="application" xlink:href="elife-67806-supp12-v1.bz2"/></supplementary-material><supplementary-material id="supp13"><label>Supplementary file 13.</label><caption><title>List of DNA libraries grouped by sequencing runs.</title></caption><media mime-subtype="excel" mimetype="application" xlink:href="elife-67806-supp13-v1.xls"/></supplementary-material><supplementary-material id="supp14"><label>Supplementary file 14.</label><caption><title>List of oligonucleotides used in this study.</title></caption><media mime-subtype="excel" mimetype="application" xlink:href="elife-67806-supp14-v1.xls"/></supplementary-material><supplementary-material id="supp15"><label>Supplementary file 15.</label><caption><title>Construction of single-site mutant strains.</title></caption><media mime-subtype="excel" mimetype="application" xlink:href="elife-67806-supp15-v1.xls"/></supplementary-material><supplementary-material id="supp16"><label>Supplementary file 16.</label><caption><title>Phenotypes of RAP1 mutants (expression) and GCR1 mutants (expression and fitness).</title></caption><media mime-subtype="excel" mimetype="application" xlink:href="elife-67806-supp16-v1.xls"/></supplementary-material><supplementary-material id="transrepform"><label>Transparent reporting form</label><media mime-subtype="docx" mimetype="application" xlink:href="elife-67806-transrepform-v1.docx"/></supplementary-material></sec><sec id="s7" sec-type="data-availability"><title>Data availability</title><p>Sequencing data have been deposited in NCBI SRA under BioProject code PRJNA706682. Flow cytometry data have been deposited in the FlowRepository (<ext-link ext-link-type="uri" xlink:href="https://flowrepository.org/">https://flowrepository.org/</ext-link>) under experiments ID FR-FCM-Z3WV, FR-FCM-Z3JY, FR-FCM-Z3J2, FR-FCM-Z3J3 and FR-FCM-Z3J5. All data generated or analysed during this study are included in Supplementary Files. Source data have been provided in Supplementary File 20 for Figure 1B-E, Figure 2E-G, Figure 3A,B,E, Figure 5C-G, Figure 6, Figure 7, Figure 2 - figure supplements 1-5, Figure 3 - figure supplements 1-2.</p><p>The following datasets were generated:</p><p><element-citation id="dataset1" publication-type="data" specific-use="isSupplementedBy"><person-group person-group-type="author"><name><surname>Duveau</surname><given-names>F</given-names></name><name><surname>Wittkopp</surname><given-names>PJ</given-names></name></person-group><year iso-8601-date="2021">2021</year><data-title>Mapping yeast trans-regulatory mutations by sequencing bulk segregant populations</data-title><source>NCBI BioProject</source><pub-id assigning-authority="NCBI" pub-id-type="accession" xlink:href="https://www.ncbi.nlm.nih.gov/bioproject/PRJNA706682">PRJNA706682</pub-id></element-citation></p><p><element-citation id="dataset2" publication-type="data" specific-use="isSupplementedBy"><person-group person-group-type="author"><name><surname>Duveau</surname><given-names>F</given-names></name><name><surname>Wittkopp</surname><given-names>PJ</given-names></name></person-group><year iso-8601-date="2021">2021</year><data-title>Secondary screen of pTDH3-YFP expression changes in yeast EMS mutants</data-title><source>FlowRepository</source><pub-id assigning-authority="other" pub-id-type="accession" xlink:href="http://flowrepository.org/id/FR-FCM-Z3WV">FR-FCM-Z3WV</pub-id></element-citation></p><p><element-citation id="dataset3" publication-type="data" specific-use="isSupplementedBy"><person-group person-group-type="author"><name><surname>Duveau</surname><given-names>F</given-names></name><name><surname>Wittkopp</surname><given-names>PJ</given-names></name></person-group><year iso-8601-date="2021">2021</year><data-title>Expression of pTDH3-YFP in trans-regulatory mutant strains of yeast</data-title><source>FlowRepository</source><pub-id assigning-authority="other" pub-id-type="accession" xlink:href="http://flowrepository.org/id/FR-FCM-Z3JY">FR-FCM-Z3JY</pub-id></element-citation></p><p><element-citation id="dataset4" publication-type="data" specific-use="isSupplementedBy"><person-group person-group-type="author"><name><surname>Duveau</surname><given-names>F</given-names></name><name><surname>Wittkopp</surname><given-names>PJ</given-names></name></person-group><year iso-8601-date="2021">2021</year><data-title>Expression of pTDH3-YFP in RAP1 mutant strains of yeast</data-title><source>FlowRepository</source><pub-id assigning-authority="other" pub-id-type="accession" xlink:href="http://flowrepository.org/id/FR-FCM-Z3J2">FR-FCM-Z3J2</pub-id></element-citation></p><p><element-citation id="dataset5" publication-type="data" specific-use="isSupplementedBy"><person-group person-group-type="author"><name><surname>Duveau</surname><given-names>F</given-names></name><name><surname>Wittkopp</surname><given-names>PJ</given-names></name></person-group><year iso-8601-date="2021">2021</year><data-title>Expression of pTDH3-YFP in GCR1 mutant strains of yeast</data-title><source>FlowRepository</source><pub-id assigning-authority="other" pub-id-type="accession" xlink:href="http://flowrepository.org/id/FR-FCM-Z3J3">FR-FCM-Z3J3</pub-id></element-citation></p><p><element-citation id="dataset6" publication-type="data" specific-use="isSupplementedBy"><person-group person-group-type="author"><name><surname>Duveau</surname><given-names>F</given-names></name><name><surname>Wittkopp</surname><given-names>PJ</given-names></name></person-group><year iso-8601-date="2021">2021</year><data-title>Fitness of GCR1 mutant strains of yeast in rich medium</data-title><source>FlowRepository</source><pub-id assigning-authority="Dryad" pub-id-type="accession" xlink:href="http://flowrepository.org/id/FR-FCM-Z3J5">FR-FCM-Z3J5</pub-id></element-citation></p></sec><ref-list><title>References</title><ref id="bib1"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Albert</surname> <given-names>FW</given-names></name><name><surname>Bloom</surname> <given-names>JS</given-names></name><name><surname>Siegel</surname> <given-names>J</given-names></name><name><surname>Day</surname> <given-names>L</given-names></name><name><surname>Kruglyak</surname> <given-names>L</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Genetics of <italic>trans</italic>-regulatory variation in gene expression</article-title><source>eLife</source><volume>7</volume><elocation-id>e35471</elocation-id><pub-id pub-id-type="doi">10.7554/eLife.35471</pub-id></element-citation></ref><ref id="bib2"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Albert</surname> <given-names>FW</given-names></name><name><surname>Kruglyak</surname> <given-names>L</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>The role of regulatory variation in complex traits and disease</article-title><source>Nature Reviews Genetics</source><volume>16</volume><fpage>197</fpage><lpage>212</lpage><pub-id pub-id-type="doi">10.1038/nrg3891</pub-id></element-citation></ref><ref id="bib3"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Barbeira</surname> <given-names>AN</given-names></name><name><surname>Dickinson</surname> <given-names>SP</given-names></name><name><surname>Bonazzola</surname> <given-names>R</given-names></name><name><surname>Zheng</surname> <given-names>J</given-names></name><name><surname>Wheeler</surname> <given-names>HE</given-names></name><name><surname>Torres</surname> <given-names>JM</given-names></name><name><surname>Torstenson</surname> <given-names>ES</given-names></name><name><surname>Shah</surname> <given-names>KP</given-names></name><name><surname>Garcia</surname> <given-names>T</given-names></name><name><surname>Edwards</surname> <given-names>TL</given-names></name><name><surname>Stahl</surname> <given-names>EA</given-names></name><name><surname>Huckins</surname> <given-names>LM</given-names></name><name><surname>Nicolae</surname> <given-names>DL</given-names></name><name><surname>Cox</surname> <given-names>NJ</given-names></name><name><surname>Im</surname> <given-names>HK</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Exploring the phenotypic consequences of tissue specific gene expression variation inferred from GWAS summary statistics</article-title><source>Nature Communications</source><volume>9</volume><elocation-id>1825</elocation-id><pub-id pub-id-type="doi">10.1038/s41467-018-03621-1</pub-id></element-citation></ref><ref id="bib4"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Bhate</surname> <given-names>M</given-names></name><name><surname>Wang</surname> <given-names>X</given-names></name><name><surname>Baum</surname> <given-names>J</given-names></name><name><surname>Brodsky</surname> <given-names>B</given-names></name></person-group><year iso-8601-date="2002">2002</year><article-title>Folding and conformational consequences of glycine to alanine replacements at different positions in a collagen model peptide</article-title><source>Biochemistry</source><volume>41</volume><fpage>6539</fpage><lpage>6547</lpage><pub-id pub-id-type="doi">10.1021/bi020070d</pub-id><pub-id pub-id-type="pmid">12009919</pub-id></element-citation></ref><ref id="bib5"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Chin</surname> <given-names>BL</given-names></name><name><surname>Frizzell</surname> <given-names>MA</given-names></name><name><surname>Timberlake</surname> <given-names>WE</given-names></name><name><surname>Fink</surname> <given-names>GR</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>FASTER MT: isolation of pure populations of a and α ascospores from Saccharomyces cerevisiae</article-title><source>G3: Genes, Genomes, Genetics</source><volume>2</volume><fpage>449</fpage><lpage>452</lpage><pub-id pub-id-type="doi">10.1534/g3.111.001826</pub-id></element-citation></ref><ref id="bib6"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Clifton</surname> <given-names>D</given-names></name><name><surname>Weinstock</surname> <given-names>SB</given-names></name><name><surname>Fraenkel</surname> <given-names>DG</given-names></name></person-group><year iso-8601-date="1978">1978</year><article-title>Glycolysis mutants in <italic>Saccharomyces</italic> cerevisiae</article-title><source>Genetics</source><volume>88</volume><fpage>1</fpage><lpage>11</lpage><pub-id pub-id-type="pmid">147195</pub-id></element-citation></ref><ref id="bib7"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Costanzo</surname> <given-names>M</given-names></name><name><surname>VanderSluis</surname> <given-names>B</given-names></name><name><surname>Koch</surname> <given-names>EN</given-names></name><name><surname>Baryshnikova</surname> <given-names>A</given-names></name><name><surname>Pons</surname> <given-names>C</given-names></name><name><surname>Tan</surname> <given-names>G</given-names></name><name><surname>Wang</surname> <given-names>W</given-names></name><name><surname>Usaj</surname> <given-names>M</given-names></name><name><surname>Hanchard</surname> <given-names>J</given-names></name><name><surname>Lee</surname> <given-names>SD</given-names></name><name><surname>Pelechano</surname> <given-names>V</given-names></name><name><surname>Styles</surname> <given-names>EB</given-names></name><name><surname>Billmann</surname> <given-names>M</given-names></name><name><surname>van Leeuwen</surname> <given-names>J</given-names></name><name><surname>van Dyk</surname> <given-names>N</given-names></name><name><surname>Lin</surname> <given-names>ZY</given-names></name><name><surname>Kuzmin</surname> <given-names>E</given-names></name><name><surname>Nelson</surname> <given-names>J</given-names></name><name><surname>Piotrowski</surname> <given-names>JS</given-names></name><name><surname>Srikumar</surname> <given-names>T</given-names></name><name><surname>Bahr</surname> <given-names>S</given-names></name><name><surname>Chen</surname> <given-names>Y</given-names></name><name><surname>Deshpande</surname> <given-names>R</given-names></name><name><surname>Kurat</surname> <given-names>CF</given-names></name><name><surname>Li</surname> <given-names>SC</given-names></name><name><surname>Li</surname> <given-names>Z</given-names></name><name><surname>Usaj</surname> <given-names>MM</given-names></name><name><surname>Okada</surname> <given-names>H</given-names></name><name><surname>Pascoe</surname> <given-names>N</given-names></name><name><surname>San Luis</surname> <given-names>BJ</given-names></name><name><surname>Sharifpoor</surname> <given-names>S</given-names></name><name><surname>Shuteriqi</surname> <given-names>E</given-names></name><name><surname>Simpkins</surname> <given-names>SW</given-names></name><name><surname>Snider</surname> <given-names>J</given-names></name><name><surname>Suresh</surname> <given-names>HG</given-names></name><name><surname>Tan</surname> <given-names>Y</given-names></name><name><surname>Zhu</surname> <given-names>H</given-names></name><name><surname>Malod-Dognin</surname> <given-names>N</given-names></name><name><surname>Janjic</surname> <given-names>V</given-names></name><name><surname>Przulj</surname> <given-names>N</given-names></name><name><surname>Troyanskaya</surname> <given-names>OG</given-names></name><name><surname>Stagljar</surname> <given-names>I</given-names></name><name><surname>Xia</surname> <given-names>T</given-names></name><name><surname>Ohya</surname> <given-names>Y</given-names></name><name><surname>Gingras</surname> <given-names>AC</given-names></name><name><surname>Raught</surname> <given-names>B</given-names></name><name><surname>Boutros</surname> <given-names>M</given-names></name><name><surname>Steinmetz</surname> <given-names>LM</given-names></name><name><surname>Moore</surname> <given-names>CL</given-names></name><name><surname>Rosebrock</surname> <given-names>AP</given-names></name><name><surname>Caudy</surname> <given-names>AA</given-names></name><name><surname>Myers</surname> <given-names>CL</given-names></name><name><surname>Andrews</surname> <given-names>B</given-names></name><name><surname>Boone</surname> <given-names>C</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>A global genetic interaction network maps a wiring diagram of cellular function</article-title><source>Science</source><volume>353</volume><elocation-id>aaf1420</elocation-id><pub-id pub-id-type="doi">10.1126/science.aaf1420</pub-id><pub-id pub-id-type="pmid">27708008</pub-id></element-citation></ref><ref id="bib8"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Deutschbauer</surname> <given-names>AM</given-names></name><name><surname>Davis</surname> <given-names>RW</given-names></name></person-group><year iso-8601-date="2005">2005</year><article-title>Quantitative trait loci mapped to single-nucleotide resolution in yeast</article-title><source>Nature Genetics</source><volume>37</volume><fpage>1333</fpage><lpage>1340</lpage><pub-id pub-id-type="doi">10.1038/ng1674</pub-id></element-citation></ref><ref id="bib9"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Dimitrov</surname> <given-names>LN</given-names></name><name><surname>Brem</surname> <given-names>RB</given-names></name><name><surname>Kruglyak</surname> <given-names>L</given-names></name><name><surname>Gottschling</surname> <given-names>DE</given-names></name></person-group><year iso-8601-date="2009">2009</year><article-title>Polymorphisms in multiple genes contribute to the spontaneous mitochondrial genome instability of <italic>Saccharomyces cerevisiae</italic> S288C strains</article-title><source>Genetics</source><volume>183</volume><fpage>365</fpage><lpage>383</lpage><pub-id pub-id-type="doi">10.1534/genetics.109.104497</pub-id><pub-id pub-id-type="pmid">19581448</pub-id></element-citation></ref><ref id="bib10"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Duveau</surname> <given-names>F</given-names></name><name><surname>Metzger</surname> <given-names>BPH</given-names></name><name><surname>Gruber</surname> <given-names>JD</given-names></name><name><surname>Mack</surname> <given-names>K</given-names></name><name><surname>Sood</surname> <given-names>N</given-names></name><name><surname>Brooks</surname> <given-names>TE</given-names></name><name><surname>Wittkopp</surname> <given-names>PJ</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>Mapping small effect mutations in <italic>Saccharomyces cerevisiae</italic>: impacts of experimental design and mutational properties</article-title><source>G3: Genes, Genomes, Genetics</source><volume>4</volume><fpage>1205</fpage><lpage>1216</lpage><pub-id pub-id-type="doi">10.1534/g3.114.011783</pub-id></element-citation></ref><ref id="bib11"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Duveau</surname> <given-names>F</given-names></name><name><surname>Hodgins-Davis</surname> <given-names>A</given-names></name><name><surname>Metzger</surname> <given-names>BP</given-names></name><name><surname>Yang</surname> <given-names>B</given-names></name><name><surname>Tryban</surname> <given-names>S</given-names></name><name><surname>Walker</surname> <given-names>EA</given-names></name><name><surname>Lybrook</surname> <given-names>T</given-names></name><name><surname>Wittkopp</surname> <given-names>PJ</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Fitness effects of altering gene expression noise in <italic>Saccharomyces cerevisiae</italic></article-title><source>eLife</source><volume>7</volume><elocation-id>e37272</elocation-id><pub-id pub-id-type="doi">10.7554/eLife.37272</pub-id><pub-id pub-id-type="pmid">30124429</pub-id></element-citation></ref><ref id="bib12"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Featherstone</surname> <given-names>DE</given-names></name><name><surname>Broadie</surname> <given-names>K</given-names></name></person-group><year iso-8601-date="2002">2002</year><article-title>Wrestling with pleiotropy: genomic and topological analysis of the yeast gene expression network</article-title><source>BioEssays</source><volume>24</volume><fpage>267</fpage><lpage>274</lpage><pub-id pub-id-type="doi">10.1002/bies.10054</pub-id><pub-id pub-id-type="pmid">11891763</pub-id></element-citation></ref><ref id="bib13"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ferraro</surname> <given-names>NM</given-names></name><name><surname>Strober</surname> <given-names>BJ</given-names></name><name><surname>Einson</surname> <given-names>J</given-names></name><name><surname>Abell</surname> <given-names>NS</given-names></name><name><surname>Aguet</surname> <given-names>F</given-names></name><name><surname>Barbeira</surname> <given-names>AN</given-names></name><name><surname>Brandt</surname> <given-names>M</given-names></name><name><surname>Bucan</surname> <given-names>M</given-names></name><name><surname>Castel</surname> <given-names>SE</given-names></name><name><surname>Davis</surname> <given-names>JR</given-names></name><name><surname>Greenwald</surname> <given-names>E</given-names></name><name><surname>Hess</surname> <given-names>GT</given-names></name><name><surname>Hilliard</surname> <given-names>AT</given-names></name><name><surname>Kember</surname> <given-names>RL</given-names></name><name><surname>Kotis</surname> <given-names>B</given-names></name><name><surname>Park</surname> <given-names>Y</given-names></name><name><surname>Peloso</surname> <given-names>G</given-names></name><name><surname>Ramdas</surname> <given-names>S</given-names></name><name><surname>Scott</surname> <given-names>AJ</given-names></name><name><surname>Smail</surname> <given-names>C</given-names></name><name><surname>Tsang</surname> <given-names>EK</given-names></name><name><surname>Zekavat</surname> <given-names>SM</given-names></name><name><surname>Ziosi</surname> <given-names>M</given-names></name><name><surname>Aradhana</surname></name> <name><surname>Ardlie</surname> <given-names>KG</given-names></name><name><surname>Assimes</surname> <given-names>TL</given-names></name><name><surname>Bassik</surname> <given-names>MC</given-names></name><name><surname>Brown</surname> <given-names>CD</given-names></name><name><surname>Correa</surname> <given-names>A</given-names></name><name><surname>Hall</surname> <given-names>I</given-names></name><name><surname>Im</surname> <given-names>HK</given-names></name><name><surname>Li</surname> <given-names>X</given-names></name><name><surname>Natarajan</surname> <given-names>P</given-names></name><name><surname>Lappalainen</surname> <given-names>T</given-names></name><name><surname>Mohammadi</surname> <given-names>P</given-names></name><name><surname>Montgomery</surname> <given-names>SB</given-names></name><name><surname>Battle</surname> <given-names>A</given-names></name><collab>TOPMed Lipids Working Group</collab><collab>GTEx Consortium</collab></person-group><year iso-8601-date="2020">2020</year><article-title>Transcriptomic signatures across human tissues identify functional rare genetic variation</article-title><source>Science</source><volume>369</volume><elocation-id>eaaz5900</elocation-id><pub-id pub-id-type="doi">10.1126/science.aaz5900</pub-id><pub-id pub-id-type="pmid">32913073</pub-id></element-citation></ref><ref id="bib14"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Flint</surname> <given-names>J</given-names></name><name><surname>Ideker</surname> <given-names>T</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>The great hairball gambit</article-title><source>PLOS Genetics</source><volume>15</volume><elocation-id>e1008519</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pgen.1008519</pub-id><pub-id pub-id-type="pmid">31770365</pub-id></element-citation></ref><ref id="bib15"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Gamazon</surname> <given-names>ER</given-names></name><name><surname>Segrè</surname> <given-names>AV</given-names></name><name><surname>van de Bunt</surname> <given-names>M</given-names></name><name><surname>Wen</surname> <given-names>X</given-names></name><name><surname>Xi</surname> <given-names>HS</given-names></name><name><surname>Hormozdiari</surname> <given-names>F</given-names></name><name><surname>Ongen</surname> <given-names>H</given-names></name><name><surname>Konkashbaev</surname> <given-names>A</given-names></name><name><surname>Derks</surname> <given-names>EM</given-names></name><name><surname>Aguet</surname> <given-names>F</given-names></name><name><surname>Quan</surname> <given-names>J</given-names></name><name><surname>Nicolae</surname> <given-names>DL</given-names></name><name><surname>Eskin</surname> <given-names>E</given-names></name><name><surname>Kellis</surname> <given-names>M</given-names></name><name><surname>Getz</surname> <given-names>G</given-names></name><name><surname>McCarthy</surname> <given-names>MI</given-names></name><name><surname>Dermitzakis</surname> <given-names>ET</given-names></name><name><surname>Cox</surname> <given-names>NJ</given-names></name><name><surname>Ardlie</surname> <given-names>KG</given-names></name><collab>GTEx Consortium</collab></person-group><year iso-8601-date="2018">2018</year><article-title>Using an atlas of gene regulation across 44 human tissues to inform complex disease- and trait-associated variation</article-title><source>Nature Genetics</source><volume>50</volume><fpage>956</fpage><lpage>967</lpage><pub-id pub-id-type="doi">10.1038/s41588-018-0154-4</pub-id><pub-id pub-id-type="pmid">29955180</pub-id></element-citation></ref><ref id="bib16"><element-citation publication-type="preprint"><person-group person-group-type="author"><name><surname>Garrison</surname> <given-names>E</given-names></name><name><surname>Marth</surname> <given-names>G</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>Haplotype-based variant detection from short-read sequencing</article-title><source>arXiv</source><ext-link ext-link-type="uri" xlink:href="http://arxiv.org/abs/1207.3907">http://arxiv.org/abs/1207.3907</ext-link></element-citation></ref><ref id="bib17"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Gavin</surname> <given-names>AC</given-names></name><name><surname>Bösche</surname> <given-names>M</given-names></name><name><surname>Krause</surname> <given-names>R</given-names></name><name><surname>Grandi</surname> <given-names>P</given-names></name><name><surname>Marzioch</surname> <given-names>M</given-names></name><name><surname>Bauer</surname> <given-names>A</given-names></name><name><surname>Schultz</surname> <given-names>J</given-names></name><name><surname>Rick</surname> <given-names>JM</given-names></name><name><surname>Michon</surname> <given-names>AM</given-names></name><name><surname>Cruciat</surname> <given-names>CM</given-names></name><name><surname>Remor</surname> <given-names>M</given-names></name><name><surname>Höfert</surname> <given-names>C</given-names></name><name><surname>Schelder</surname> <given-names>M</given-names></name><name><surname>Brajenovic</surname> <given-names>M</given-names></name><name><surname>Ruffner</surname> <given-names>H</given-names></name><name><surname>Merino</surname> <given-names>A</given-names></name><name><surname>Klein</surname> <given-names>K</given-names></name><name><surname>Hudak</surname> <given-names>M</given-names></name><name><surname>Dickson</surname> <given-names>D</given-names></name><name><surname>Rudi</surname> <given-names>T</given-names></name><name><surname>Gnau</surname> <given-names>V</given-names></name><name><surname>Bauch</surname> <given-names>A</given-names></name><name><surname>Bastuck</surname> <given-names>S</given-names></name><name><surname>Huhse</surname> <given-names>B</given-names></name><name><surname>Leutwein</surname> <given-names>C</given-names></name><name><surname>Heurtier</surname> <given-names>MA</given-names></name><name><surname>Copley</surname> <given-names>RR</given-names></name><name><surname>Edelmann</surname> <given-names>A</given-names></name><name><surname>Querfurth</surname> <given-names>E</given-names></name><name><surname>Rybin</surname> <given-names>V</given-names></name><name><surname>Drewes</surname> <given-names>G</given-names></name><name><surname>Raida</surname> <given-names>M</given-names></name><name><surname>Bouwmeester</surname> <given-names>T</given-names></name><name><surname>Bork</surname> <given-names>P</given-names></name><name><surname>Seraphin</surname> <given-names>B</given-names></name><name><surname>Kuster</surname> <given-names>B</given-names></name><name><surname>Neubauer</surname> <given-names>G</given-names></name><name><surname>Superti-Furga</surname> <given-names>G</given-names></name></person-group><year iso-8601-date="2002">2002</year><article-title>Functional organization of the yeast proteome by systematic analysis of protein complexes</article-title><source>Nature</source><volume>415</volume><fpage>141</fpage><lpage>147</lpage><pub-id pub-id-type="doi">10.1038/415141a</pub-id><pub-id pub-id-type="pmid">11805826</pub-id></element-citation></ref><ref id="bib18"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ghaemmaghami</surname> <given-names>S</given-names></name><name><surname>Huh</surname> <given-names>WK</given-names></name><name><surname>Bower</surname> <given-names>K</given-names></name><name><surname>Howson</surname> <given-names>RW</given-names></name><name><surname>Belle</surname> <given-names>A</given-names></name><name><surname>Dephoure</surname> <given-names>N</given-names></name><name><surname>O'Shea</surname> <given-names>EK</given-names></name><name><surname>Weissman</surname> <given-names>JS</given-names></name></person-group><year iso-8601-date="2003">2003</year><article-title>Global analysis of protein expression in yeast</article-title><source>Nature</source><volume>425</volume><fpage>737</fpage><lpage>741</lpage><pub-id pub-id-type="doi">10.1038/nature02046</pub-id><pub-id pub-id-type="pmid">14562106</pub-id></element-citation></ref><ref id="bib19"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Gietz</surname> <given-names>RD</given-names></name><name><surname>Schiestl</surname> <given-names>RH</given-names></name></person-group><year iso-8601-date="2007">2007</year><article-title>High-efficiency yeast transformation using the LiAc/SS carrier DNA/PEG method</article-title><source>Nature Protocols</source><volume>2</volume><fpage>31</fpage><lpage>34</lpage><pub-id pub-id-type="doi">10.1038/nprot.2007.13</pub-id><pub-id pub-id-type="pmid">17401334</pub-id></element-citation></ref><ref id="bib20"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Gruber</surname> <given-names>JD</given-names></name><name><surname>Vogel</surname> <given-names>K</given-names></name><name><surname>Kalay</surname> <given-names>G</given-names></name><name><surname>Wittkopp</surname> <given-names>PJ</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>Contrasting properties of Gene-Specific regulatory, coding, and copy number mutations in <italic>Saccharomyces cerevisiae</italic>: Frequency, Effects, and Dominance</article-title><source>PLOS Genetics</source><volume>8</volume><elocation-id>e1002497</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pgen.1002497</pub-id></element-citation></ref><ref id="bib21"><element-citation publication-type="journal"><person-group person-group-type="author"><collab>GTEx Consortium</collab></person-group><year iso-8601-date="2020">2020</year><article-title>The GTEx consortium atlas of genetic regulatory effects across human tissues</article-title><source>Science</source><volume>369</volume><fpage>1318</fpage><lpage>1330</lpage><pub-id pub-id-type="doi">10.1126/science.aaz1776</pub-id><pub-id pub-id-type="pmid">32913098</pub-id></element-citation></ref><ref id="bib22"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hill</surname> <given-names>MS</given-names></name><name><surname>Vande Zande</surname> <given-names>P</given-names></name><name><surname>Wittkopp</surname> <given-names>PJ</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>Molecular and evolutionary processes generating variation in gene expression</article-title><source>Nature Reviews Genetics</source><volume>22</volume><fpage>203</fpage><lpage>215</lpage><pub-id pub-id-type="doi">10.1038/s41576-020-00304-w</pub-id><pub-id pub-id-type="pmid">33268840</pub-id></element-citation></ref><ref id="bib23"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hodgins-Davis</surname> <given-names>A</given-names></name><name><surname>Duveau</surname> <given-names>F</given-names></name><name><surname>Walker</surname> <given-names>EA</given-names></name><name><surname>Wittkopp</surname> <given-names>PJ</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Empirical measures of mutational effects define neutral models of regulatory evolution in <italic>Saccharomyces cerevisiae</italic></article-title><source>PNAS</source><volume>116</volume><fpage>21085</fpage><lpage>21093</lpage><pub-id pub-id-type="doi">10.1073/pnas.1902823116</pub-id><pub-id pub-id-type="pmid">31570626</pub-id></element-citation></ref><ref id="bib24"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hornung</surname> <given-names>G</given-names></name><name><surname>Oren</surname> <given-names>M</given-names></name><name><surname>Barkai</surname> <given-names>N</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>Nucleosome organization affects the sensitivity of gene expression to promoter mutations</article-title><source>Molecular Cell</source><volume>46</volume><fpage>362</fpage><lpage>368</lpage><pub-id pub-id-type="doi">10.1016/j.molcel.2012.02.019</pub-id><pub-id pub-id-type="pmid">22464732</pub-id></element-citation></ref><ref id="bib25"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hossain</surname> <given-names>MA</given-names></name><name><surname>Claggett</surname> <given-names>JM</given-names></name><name><surname>Edwards</surname> <given-names>SR</given-names></name><name><surname>Shi</surname> <given-names>A</given-names></name><name><surname>Pennebaker</surname> <given-names>SL</given-names></name><name><surname>Cheng</surname> <given-names>MY</given-names></name><name><surname>Hasty</surname> <given-names>J</given-names></name><name><surname>Johnson</surname> <given-names>TL</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Posttranscriptional regulation of Gcr1 expression and activity is crucial for metabolic adjustment in response to glucose availability</article-title><source>Molecular Cell</source><volume>62</volume><fpage>346</fpage><lpage>358</lpage><pub-id pub-id-type="doi">10.1016/j.molcel.2016.04.012</pub-id><pub-id pub-id-type="pmid">27153533</pub-id></element-citation></ref><ref id="bib26"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hughes</surname> <given-names>TR</given-names></name><name><surname>Marton</surname> <given-names>MJ</given-names></name><name><surname>Jones</surname> <given-names>AR</given-names></name><name><surname>Roberts</surname> <given-names>CJ</given-names></name><name><surname>Stoughton</surname> <given-names>R</given-names></name><name><surname>Armour</surname> <given-names>CD</given-names></name><name><surname>Bennett</surname> <given-names>HA</given-names></name><name><surname>Coffey</surname> <given-names>E</given-names></name><name><surname>Dai</surname> <given-names>H</given-names></name><name><surname>He</surname> <given-names>YD</given-names></name><name><surname>Kidd</surname> <given-names>MJ</given-names></name><name><surname>King</surname> <given-names>AM</given-names></name><name><surname>Meyer</surname> <given-names>MR</given-names></name><name><surname>Slade</surname> <given-names>D</given-names></name><name><surname>Lum</surname> <given-names>PY</given-names></name><name><surname>Stepaniants</surname> <given-names>SB</given-names></name><name><surname>Shoemaker</surname> <given-names>DD</given-names></name><name><surname>Gachotte</surname> <given-names>D</given-names></name><name><surname>Chakraburtty</surname> <given-names>K</given-names></name><name><surname>Simon</surname> <given-names>J</given-names></name><name><surname>Bard</surname> <given-names>M</given-names></name><name><surname>Friend</surname> <given-names>SH</given-names></name></person-group><year iso-8601-date="2000">2000</year><article-title>Functional discovery via a compendium of expression profiles</article-title><source>Cell</source><volume>102</volume><fpage>109</fpage><lpage>126</lpage><pub-id pub-id-type="doi">10.1016/S0092-8674(00)00015-5</pub-id></element-citation></ref><ref id="bib27"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hughes</surname> <given-names>TR</given-names></name><name><surname>de Boer</surname> <given-names>CG</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>Mapping yeast transcriptional networks</article-title><source>Genetics</source><volume>195</volume><fpage>9</fpage><lpage>36</lpage><pub-id pub-id-type="doi">10.1534/genetics.113.153262</pub-id><pub-id pub-id-type="pmid">24018767</pub-id></element-citation></ref><ref id="bib28"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Huie</surname> <given-names>MA</given-names></name><name><surname>Scott</surname> <given-names>EW</given-names></name><name><surname>Drazinic</surname> <given-names>CM</given-names></name><name><surname>Lopez</surname> <given-names>MC</given-names></name><name><surname>Hornstra</surname> <given-names>IK</given-names></name><name><surname>Yang</surname> <given-names>TP</given-names></name><name><surname>Baker</surname> <given-names>HV</given-names></name></person-group><year iso-8601-date="1992">1992</year><article-title>Characterization of the DNA-binding activity of GCR1: in vivo evidence for two GCR1-binding sites in the upstream activating sequence of TPI of Saccharomyces cerevisiae</article-title><source>Molecular and Cellular Biology</source><volume>12</volume><fpage>2690</fpage><lpage>2700</lpage><pub-id pub-id-type="doi">10.1128/MCB.12.6.2690</pub-id><pub-id pub-id-type="pmid">1588965</pub-id></element-citation></ref><ref id="bib29"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Jackson</surname> <given-names>CA</given-names></name><name><surname>Castro</surname> <given-names>DM</given-names></name><name><surname>Saldi</surname> <given-names>GA</given-names></name><name><surname>Bonneau</surname> <given-names>R</given-names></name><name><surname>Gresham</surname> <given-names>D</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>Gene regulatory network reconstruction using single-cell RNA sequencing of barcoded genotypes in diverse environments</article-title><source>eLife</source><volume>9</volume><elocation-id>e51254</elocation-id><pub-id pub-id-type="doi">10.7554/eLife.51254</pub-id><pub-id pub-id-type="pmid">31985403</pub-id></element-citation></ref><ref id="bib30"><element-citation publication-type="software"><person-group person-group-type="author"><name><surname>Joshi</surname> <given-names>NA</given-names></name><name><surname>Fass</surname> <given-names>JN</given-names></name></person-group><year iso-8601-date="2011">2011</year><data-title>Sickle: A sliding-window, adaptive, quality-based trimming tool for FastQ files</data-title><source>Github</source><version designator="1.33">1.33</version><ext-link ext-link-type="uri" xlink:href="https://github.com/najoshi/sickle">https://github.com/najoshi/sickle</ext-link></element-citation></ref><ref id="bib31"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kafri</surname> <given-names>M</given-names></name><name><surname>Metzl-Raz</surname> <given-names>E</given-names></name><name><surname>Jona</surname> <given-names>G</given-names></name><name><surname>Barkai</surname> <given-names>N</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>The cost of protein production</article-title><source>Cell Reports</source><volume>14</volume><fpage>22</fpage><lpage>31</lpage><pub-id pub-id-type="doi">10.1016/j.celrep.2015.12.015</pub-id><pub-id pub-id-type="pmid">26725116</pub-id></element-citation></ref><ref id="bib32"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kemmeren</surname> <given-names>P</given-names></name><name><surname>Sameith</surname> <given-names>K</given-names></name><name><surname>van de Pasch</surname> <given-names>LA</given-names></name><name><surname>Benschop</surname> <given-names>JJ</given-names></name><name><surname>Lenstra</surname> <given-names>TL</given-names></name><name><surname>Margaritis</surname> <given-names>T</given-names></name><name><surname>O'Duibhir</surname> <given-names>E</given-names></name><name><surname>Apweiler</surname> <given-names>E</given-names></name><name><surname>van Wageningen</surname> <given-names>S</given-names></name><name><surname>Ko</surname> <given-names>CW</given-names></name><name><surname>van Heesch</surname> <given-names>S</given-names></name><name><surname>Kashani</surname> <given-names>MM</given-names></name><name><surname>Ampatziadis-Michailidis</surname> <given-names>G</given-names></name><name><surname>Brok</surname> <given-names>MO</given-names></name><name><surname>Brabers</surname> <given-names>NA</given-names></name><name><surname>Miles</surname> <given-names>AJ</given-names></name><name><surname>Bouwmeester</surname> <given-names>D</given-names></name><name><surname>van Hooff</surname> <given-names>SR</given-names></name><name><surname>van Bakel</surname> <given-names>H</given-names></name><name><surname>Sluiters</surname> <given-names>E</given-names></name><name><surname>Bakker</surname> <given-names>LV</given-names></name><name><surname>Snel</surname> <given-names>B</given-names></name><name><surname>Lijnzaad</surname> <given-names>P</given-names></name><name><surname>van Leenen</surname> <given-names>D</given-names></name><name><surname>Groot Koerkamp</surname> <given-names>MJ</given-names></name><name><surname>Holstege</surname> <given-names>FC</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>Large-scale genetic perturbations reveal regulatory networks and an abundance of gene-specific repressors</article-title><source>Cell</source><volume>157</volume><fpage>740</fpage><lpage>752</lpage><pub-id pub-id-type="doi">10.1016/j.cell.2014.02.054</pub-id><pub-id pub-id-type="pmid">24766815</pub-id></element-citation></ref><ref id="bib33"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Khan</surname> <given-names>S</given-names></name><name><surname>Vihinen</surname> <given-names>M</given-names></name></person-group><year iso-8601-date="2007">2007</year><article-title>Spectrum of disease-causing mutations in protein secondary structures</article-title><source>BMC Structural Biology</source><volume>7</volume><elocation-id>56</elocation-id><pub-id pub-id-type="doi">10.1186/1472-6807-7-56</pub-id><pub-id pub-id-type="pmid">17727703</pub-id></element-citation></ref><ref id="bib34"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Konig</surname> <given-names>P</given-names></name><name><surname>Giraldo</surname> <given-names>R</given-names></name><name><surname>Chapman</surname> <given-names>L</given-names></name><name><surname>Rhodes</surname> <given-names>D</given-names></name></person-group><year iso-8601-date="1996">1996</year><article-title>The crystal structure of the DNA-binding domain of yeast RAP1 in complex with telomeric DNA</article-title><source>Cell</source><volume>85</volume><fpage>125</fpage><lpage>136</lpage><pub-id pub-id-type="doi">10.1016/S0092-8674(00)81088-0</pub-id><pub-id pub-id-type="pmid">8620531</pub-id></element-citation></ref><ref id="bib35"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kristiansson</surname> <given-names>E</given-names></name><name><surname>Thorsen</surname> <given-names>M</given-names></name><name><surname>Tamás</surname> <given-names>MJ</given-names></name><name><surname>Nerman</surname> <given-names>O</given-names></name></person-group><year iso-8601-date="2009">2009</year><article-title>Evolutionary forces act on promoter length: identification of enriched cis-regulatory elements</article-title><source>Molecular Biology and Evolution</source><volume>26</volume><fpage>1299</fpage><lpage>1307</lpage><pub-id pub-id-type="doi">10.1093/molbev/msp040</pub-id><pub-id pub-id-type="pmid">19258451</pub-id></element-citation></ref><ref id="bib36"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kwasnieski</surname> <given-names>JC</given-names></name><name><surname>Mogno</surname> <given-names>I</given-names></name><name><surname>Myers</surname> <given-names>CA</given-names></name><name><surname>Corbo</surname> <given-names>JC</given-names></name><name><surname>Cohen</surname> <given-names>BA</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>Complex effects of nucleotide variants in a mammalian <italic>cis</italic>-regulatory element</article-title><source>PNAS</source><volume>109</volume><fpage>19498</fpage><lpage>19503</lpage><pub-id pub-id-type="doi">10.1073/pnas.1210678109</pub-id><pub-id pub-id-type="pmid">23129659</pub-id></element-citation></ref><ref id="bib37"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Langmead</surname> <given-names>B</given-names></name><name><surname>Salzberg</surname> <given-names>SL</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>Fast gapped-read alignment with bowtie 2</article-title><source>Nature Methods</source><volume>9</volume><fpage>357</fpage><lpage>359</lpage><pub-id pub-id-type="doi">10.1038/nmeth.1923</pub-id><pub-id pub-id-type="pmid">22388286</pub-id></element-citation></ref><ref id="bib38"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Laughery</surname> <given-names>MF</given-names></name><name><surname>Hunter</surname> <given-names>T</given-names></name><name><surname>Brown</surname> <given-names>A</given-names></name><name><surname>Hoopes</surname> <given-names>J</given-names></name><name><surname>Ostbye</surname> <given-names>T</given-names></name><name><surname>Shumaker</surname> <given-names>T</given-names></name><name><surname>Wyrick</surname> <given-names>JJ</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>New vectors for simple and streamlined CRISPR-Cas9 genome editing in <italic>Saccharomyces cerevisiae</italic></article-title><source>Yeast</source><volume>32</volume><fpage>711</fpage><lpage>720</lpage><pub-id pub-id-type="doi">10.1002/yea.3098</pub-id><pub-id pub-id-type="pmid">26305040</pub-id></element-citation></ref><ref id="bib39"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Lewis</surname> <given-names>JA</given-names></name><name><surname>Broman</surname> <given-names>AT</given-names></name><name><surname>Will</surname> <given-names>J</given-names></name><name><surname>Gasch</surname> <given-names>AP</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>Genetic architecture of ethanol-responsive transcriptome variation in <italic>Saccharomyces cerevisiae</italic> strains</article-title><source>Genetics</source><volume>198</volume><fpage>369</fpage><lpage>382</lpage><pub-id pub-id-type="doi">10.1534/genetics.114.167429</pub-id><pub-id pub-id-type="pmid">24970865</pub-id></element-citation></ref><ref id="bib40"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Li</surname> <given-names>B</given-names></name><name><surname>Carey</surname> <given-names>M</given-names></name><name><surname>Workman</surname> <given-names>JL</given-names></name></person-group><year iso-8601-date="2007">2007</year><article-title>The role of chromatin during transcription</article-title><source>Cell</source><volume>128</volume><fpage>707</fpage><lpage>719</lpage><pub-id pub-id-type="doi">10.1016/j.cell.2007.01.015</pub-id><pub-id pub-id-type="pmid">17320508</pub-id></element-citation></ref><ref id="bib41"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>J</given-names></name><name><surname>François</surname> <given-names>JM</given-names></name><name><surname>Capp</surname> <given-names>JP</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Gene expression noise produces Cell-to-Cell heterogeneity in eukaryotic homologous recombination rate</article-title><source>Frontiers in Genetics</source><volume>10</volume><elocation-id>475</elocation-id><pub-id pub-id-type="doi">10.3389/fgene.2019.00475</pub-id><pub-id pub-id-type="pmid">31164905</pub-id></element-citation></ref><ref id="bib42"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>Z</given-names></name><name><surname>Miller</surname> <given-names>D</given-names></name><name><surname>Li</surname> <given-names>F</given-names></name><name><surname>Liu</surname> <given-names>X</given-names></name><name><surname>Levy</surname> <given-names>SF</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>A large accessory protein interactome is rewired across environments</article-title><source>eLife</source><volume>9</volume><elocation-id>e62365</elocation-id><pub-id pub-id-type="doi">10.7554/eLife.62365</pub-id><pub-id pub-id-type="pmid">32924934</pub-id></element-citation></ref><ref id="bib43"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>López</surname> <given-names>MC</given-names></name><name><surname>Baker</surname> <given-names>HV</given-names></name></person-group><year iso-8601-date="2000">2000</year><article-title>Understanding the Growth Phenotype of the Yeast <italic>gcr1</italic> Mutant in Terms of Global Genomic Expression Patterns</article-title><source>Journal of Bacteriology</source><volume>182</volume><fpage>4970</fpage><lpage>4978</lpage><pub-id pub-id-type="doi">10.1128/JB.182.17.4970-4978.2000</pub-id></element-citation></ref><ref id="bib44"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Luscombe</surname> <given-names>NM</given-names></name><name><surname>Babu</surname> <given-names>MM</given-names></name><name><surname>Yu</surname> <given-names>H</given-names></name><name><surname>Snyder</surname> <given-names>M</given-names></name><name><surname>Teichmann</surname> <given-names>SA</given-names></name><name><surname>Gerstein</surname> <given-names>M</given-names></name></person-group><year iso-8601-date="2004">2004</year><article-title>Genomic analysis of regulatory network dynamics reveals large topological changes</article-title><source>Nature</source><volume>431</volume><fpage>308</fpage><lpage>312</lpage><pub-id pub-id-type="doi">10.1038/nature02782</pub-id><pub-id pub-id-type="pmid">15372033</pub-id></element-citation></ref><ref id="bib45"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Lutz</surname> <given-names>S</given-names></name><name><surname>Brion</surname> <given-names>C</given-names></name><name><surname>Kliebhan</surname> <given-names>M</given-names></name><name><surname>Albert</surname> <given-names>FW</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>DNA variants affecting the expression of numerous genes in <italic>trans</italic> have diverse mechanisms of action and evolutionary histories</article-title><source>PLOS Genetics</source><volume>15</volume><elocation-id>e1008375</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pgen.1008375</pub-id><pub-id pub-id-type="pmid">31738765</pub-id></element-citation></ref><ref id="bib46"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Maricque</surname> <given-names>BB</given-names></name><name><surname>Dougherty</surname> <given-names>JD</given-names></name><name><surname>Cohen</surname> <given-names>BA</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>A genome-integrated massively parallel reporter assay reveals DNA sequence determinants of <italic>cis</italic>-regulatory activity in neural cells</article-title><source>Nucleic Acids Research</source><volume>45</volume><elocation-id>e16</elocation-id><pub-id pub-id-type="doi">10.1093/nar/gkw942</pub-id><pub-id pub-id-type="pmid">28204611</pub-id></element-citation></ref><ref id="bib47"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Martin</surname> <given-names>M</given-names></name></person-group><year iso-8601-date="2011">2011</year><article-title>Cutadapt removes adapter sequences from high-throughput sequencing reads</article-title><source>EMBnet Journal</source><volume>17</volume><fpage>10</fpage><lpage>12</lpage><pub-id pub-id-type="doi">10.14806/ej.17.1.200</pub-id></element-citation></ref><ref id="bib48"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>McAlister</surname> <given-names>L</given-names></name><name><surname>Holland</surname> <given-names>MJ</given-names></name></person-group><year iso-8601-date="1985">1985</year><article-title>Isolation and characterization of yeast strains carrying mutations in the glyceraldehyde-3-phosphate dehydrogenase genes</article-title><source>Journal of Biological Chemistry</source><volume>260</volume><fpage>15013</fpage><lpage>15018</lpage><pub-id pub-id-type="doi">10.1016/S0021-9258(18)95695-4</pub-id></element-citation></ref><ref id="bib49"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>McInerney</surname> <given-names>P</given-names></name><name><surname>Adams</surname> <given-names>P</given-names></name><name><surname>Hadi</surname> <given-names>MZ</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>Error rate comparison during polymerase chain reaction by DNA polymerase</article-title><source>Molecular Biology International</source><volume>2014</volume><fpage>1</fpage><lpage>8</lpage><pub-id pub-id-type="doi">10.1155/2014/287430</pub-id></element-citation></ref><ref id="bib50"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Mehrabian</surname> <given-names>M</given-names></name><name><surname>Allayee</surname> <given-names>H</given-names></name><name><surname>Stockton</surname> <given-names>J</given-names></name><name><surname>Lum</surname> <given-names>PY</given-names></name><name><surname>Drake</surname> <given-names>TA</given-names></name><name><surname>Castellani</surname> <given-names>LW</given-names></name><name><surname>Suh</surname> <given-names>M</given-names></name><name><surname>Armour</surname> <given-names>C</given-names></name><name><surname>Edwards</surname> <given-names>S</given-names></name><name><surname>Lamb</surname> <given-names>J</given-names></name><name><surname>Lusis</surname> <given-names>AJ</given-names></name><name><surname>Schadt</surname> <given-names>EE</given-names></name></person-group><year iso-8601-date="2005">2005</year><article-title>Integrating genotypic and expression data in a segregating mouse population to identify 5-lipoxygenase as a susceptibility gene for obesity and bone traits</article-title><source>Nature Genetics</source><volume>37</volume><fpage>1224</fpage><lpage>1233</lpage><pub-id pub-id-type="doi">10.1038/ng1619</pub-id><pub-id pub-id-type="pmid">16200066</pub-id></element-citation></ref><ref id="bib51"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Melnikov</surname> <given-names>A</given-names></name><name><surname>Murugan</surname> <given-names>A</given-names></name><name><surname>Zhang</surname> <given-names>X</given-names></name><name><surname>Tesileanu</surname> <given-names>T</given-names></name><name><surname>Wang</surname> <given-names>L</given-names></name><name><surname>Rogov</surname> <given-names>P</given-names></name><name><surname>Feizi</surname> <given-names>S</given-names></name><name><surname>Gnirke</surname> <given-names>A</given-names></name><name><surname>Callan</surname> <given-names>CG</given-names></name><name><surname>Kinney</surname> <given-names>JB</given-names></name><name><surname>Kellis</surname> <given-names>M</given-names></name><name><surname>Lander</surname> <given-names>ES</given-names></name><name><surname>Mikkelsen</surname> <given-names>TS</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>Systematic dissection and optimization of inducible enhancers in human cells using a massively parallel reporter assay</article-title><source>Nature Biotechnology</source><volume>30</volume><fpage>271</fpage><lpage>277</lpage><pub-id pub-id-type="doi">10.1038/nbt.2137</pub-id><pub-id pub-id-type="pmid">22371084</pub-id></element-citation></ref><ref id="bib52"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Metzger</surname> <given-names>BP</given-names></name><name><surname>Yuan</surname> <given-names>DC</given-names></name><name><surname>Gruber</surname> <given-names>JD</given-names></name><name><surname>Duveau</surname> <given-names>F</given-names></name><name><surname>Wittkopp</surname> <given-names>PJ</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>Selection on noise constrains variation in a eukaryotic promoter</article-title><source>Nature</source><volume>521</volume><fpage>344</fpage><lpage>347</lpage><pub-id pub-id-type="doi">10.1038/nature14244</pub-id><pub-id pub-id-type="pmid">25778704</pub-id></element-citation></ref><ref id="bib53"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Metzger</surname> <given-names>BP</given-names></name><name><surname>Duveau</surname> <given-names>F</given-names></name><name><surname>Yuan</surname> <given-names>DC</given-names></name><name><surname>Tryban</surname> <given-names>S</given-names></name><name><surname>Yang</surname> <given-names>B</given-names></name><name><surname>Wittkopp</surname> <given-names>PJ</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Contrasting frequencies and effects of Cis- and <italic>trans</italic>-Regulatory mutations affecting gene expression</article-title><source>Molecular Biology and Evolution</source><volume>33</volume><fpage>1131</fpage><lpage>1146</lpage><pub-id pub-id-type="doi">10.1093/molbev/msw011</pub-id><pub-id pub-id-type="pmid">26782996</pub-id></element-citation></ref><ref id="bib54"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Metzger</surname> <given-names>BPH</given-names></name><name><surname>Wittkopp</surname> <given-names>PJ</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Compensatory <italic>trans</italic> ‐regulatory alleles minimizing variation in <italic>TDH3</italic> expression are common within <italic>Saccharomyces cerevisiae</italic></article-title><source>Evolution Letters</source><volume>3</volume><fpage>448</fpage><lpage>461</lpage><pub-id pub-id-type="doi">10.1002/evl3.137</pub-id></element-citation></ref><ref id="bib55"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Mi</surname> <given-names>H</given-names></name><name><surname>Muruganujan</surname> <given-names>A</given-names></name><name><surname>Ebert</surname> <given-names>D</given-names></name><name><surname>Huang</surname> <given-names>X</given-names></name><name><surname>Thomas</surname> <given-names>PD</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>PANTHER version 14: more genomes, a new PANTHER GO-slim and improvements in enrichment analysis tools</article-title><source>Nucleic Acids Research</source><volume>47</volume><fpage>D419</fpage><lpage>D426</lpage><pub-id pub-id-type="doi">10.1093/nar/gky1038</pub-id></element-citation></ref><ref id="bib56"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Miller</surname> <given-names>BG</given-names></name></person-group><year iso-8601-date="2007">2007</year><article-title>The mutability of enzyme active-site shape determinants</article-title><source>Protein Science</source><volume>16</volume><fpage>1965</fpage><lpage>1968</lpage><pub-id pub-id-type="doi">10.1110/ps.073040307</pub-id></element-citation></ref><ref id="bib57"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Molnár</surname> <given-names>J</given-names></name><name><surname>Szakács</surname> <given-names>G</given-names></name><name><surname>Tusnády</surname> <given-names>GE</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Characterization of Disease-Associated Mutations in Human Transmembrane Proteins</article-title><source>PLOS ONE</source><volume>11</volume><elocation-id>e0151760</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pone.0151760</pub-id></element-citation></ref><ref id="bib58"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Oliver</surname> <given-names>F</given-names></name><name><surname>Christians</surname> <given-names>JK</given-names></name><name><surname>Liu</surname> <given-names>X</given-names></name><name><surname>Rhind</surname> <given-names>S</given-names></name><name><surname>Verma</surname> <given-names>V</given-names></name><name><surname>Davison</surname> <given-names>C</given-names></name><name><surname>Brown</surname> <given-names>SDM</given-names></name><name><surname>Denny</surname> <given-names>P</given-names></name><name><surname>Keightley</surname> <given-names>PD</given-names></name></person-group><year iso-8601-date="2005">2005</year><article-title>Regulatory Variation at Glypican-3 Underlies a Major Growth QTL in Mice</article-title><source>PLOS Biology</source><volume>3</volume><elocation-id>e135</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pbio.0030135</pub-id></element-citation></ref><ref id="bib59"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Outten</surname> <given-names>CE</given-names></name><name><surname>Albetel</surname> <given-names>A-N</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>Iron sensing and regulation in <italic>Saccharomyces cerevisiae</italic>: Ironing out the mechanistic details</article-title><source>Current Opinion in Microbiology</source><volume>16</volume><fpage>662</fpage><lpage>668</lpage><pub-id pub-id-type="doi">10.1016/j.mib.2013.07.020</pub-id></element-citation></ref><ref id="bib60"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Parenteau</surname> <given-names>J</given-names></name><name><surname>Maignon</surname> <given-names>L</given-names></name><name><surname>Berthoumieux</surname> <given-names>M</given-names></name><name><surname>Catala</surname> <given-names>M</given-names></name><name><surname>Gagnon</surname> <given-names>V</given-names></name><name><surname>Abou Elela</surname> <given-names>S</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Introns are mediators of cell response to starvation</article-title><source>Nature</source><volume>565</volume><fpage>612</fpage><lpage>617</lpage><pub-id pub-id-type="doi">10.1038/s41586-018-0859-7</pub-id></element-citation></ref><ref id="bib61"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Patwardhan</surname> <given-names>RP</given-names></name><name><surname>Lee</surname> <given-names>C</given-names></name><name><surname>Litvin</surname> <given-names>O</given-names></name><name><surname>Young</surname> <given-names>DL</given-names></name><name><surname>Pe'er</surname> <given-names>D</given-names></name><name><surname>Shendure</surname> <given-names>J</given-names></name></person-group><year iso-8601-date="2009">2009</year><article-title>High-resolution analysis of DNA regulatory elements by synthetic saturation mutagenesis</article-title><source>Nature Biotechnology</source><volume>27</volume><fpage>1173</fpage><lpage>1175</lpage><pub-id pub-id-type="doi">10.1038/nbt.1589</pub-id></element-citation></ref><ref id="bib62"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Piña</surname> <given-names>B</given-names></name><name><surname>Fernández-Larrea</surname> <given-names>J</given-names></name><name><surname>García-Reyero</surname> <given-names>N</given-names></name><name><surname>Idrissi</surname> <given-names>F-Z</given-names></name></person-group><year iso-8601-date="2003">2003</year><article-title>The different (sur)faces of Rap1p</article-title><source>Molecular Genetics and Genomics</source><volume>268</volume><fpage>791</fpage><lpage>798</lpage><pub-id pub-id-type="doi">10.1007/s00438-002-0801-3</pub-id></element-citation></ref><ref id="bib63"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Pinson</surname> <given-names>B</given-names></name><name><surname>Vaur</surname> <given-names>S</given-names></name><name><surname>Sagot</surname> <given-names>I</given-names></name><name><surname>Coulpier</surname> <given-names>F</given-names></name><name><surname>Lemoine</surname> <given-names>S</given-names></name><name><surname>Daignan-Fornier</surname> <given-names>B</given-names></name></person-group><year iso-8601-date="2009">2009</year><article-title>Metabolic intermediates selectively stimulate transcription factor interaction and modulate phosphate and purine pathways</article-title><source>Genes &amp; Development</source><volume>23</volume><fpage>1399</fpage><lpage>1407</lpage><pub-id pub-id-type="doi">10.1101/gad.521809</pub-id></element-citation></ref><ref id="bib64"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Rhee</surname> <given-names>HS</given-names></name><name><surname>Pugh</surname> <given-names>BF</given-names></name></person-group><year iso-8601-date="2011">2011</year><article-title>Comprehensive Genome-wide Protein-DNA Interactions Detected at Single-Nucleotide Resolution</article-title><source>Cell</source><volume>147</volume><fpage>1408</fpage><lpage>1419</lpage><pub-id pub-id-type="doi">10.1016/j.cell.2011.11.013</pub-id></element-citation></ref><ref id="bib65"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Rockman</surname> <given-names>MV</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>The qtn program and the alleles that matter for evolution: all that's gold does not glitter</article-title><source>Evolution</source><volume>66</volume><fpage>1</fpage><lpage>17</lpage><pub-id pub-id-type="doi">10.1111/j.1558-5646.2011.01486.x</pub-id></element-citation></ref><ref id="bib66"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Roman</surname> <given-names>H</given-names></name></person-group><year iso-8601-date="1956">1956</year><article-title>A system selective for mutations affecting the synthesis of Adenine in yeast compt rend trav lab</article-title><source>Carlsberg, Ser. Physiol</source><volume>26</volume><fpage>299</fpage><lpage>314</lpage></element-citation></ref><ref id="bib67"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Santangelo</surname> <given-names>GM</given-names></name></person-group><year iso-8601-date="2006">2006</year><article-title>Glucose signaling in <italic>Saccharomyces cerevisiae</italic></article-title><source>Microbiology and Molecular Biology Reviews: MMBR</source><volume>70</volume><fpage>253</fpage><lpage>282</lpage><pub-id pub-id-type="doi">10.1128/MMBR.70.1.253-282.2006</pub-id><pub-id pub-id-type="pmid">16524925</pub-id></element-citation></ref><ref id="bib68"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Schadt</surname> <given-names>EE</given-names></name><name><surname>Lamb</surname> <given-names>J</given-names></name><name><surname>Yang</surname> <given-names>X</given-names></name><name><surname>Zhu</surname> <given-names>J</given-names></name><name><surname>Edwards</surname> <given-names>S</given-names></name><name><surname>Guhathakurta</surname> <given-names>D</given-names></name><name><surname>Sieberts</surname> <given-names>SK</given-names></name><name><surname>Monks</surname> <given-names>S</given-names></name><name><surname>Reitman</surname> <given-names>M</given-names></name><name><surname>Zhang</surname> <given-names>C</given-names></name><name><surname>Lum</surname> <given-names>PY</given-names></name><name><surname>Leonardson</surname> <given-names>A</given-names></name><name><surname>Thieringer</surname> <given-names>R</given-names></name><name><surname>Metzger</surname> <given-names>JM</given-names></name><name><surname>Yang</surname> <given-names>L</given-names></name><name><surname>Castle</surname> <given-names>J</given-names></name><name><surname>Zhu</surname> <given-names>H</given-names></name><name><surname>Kash</surname> <given-names>SF</given-names></name><name><surname>Drake</surname> <given-names>TA</given-names></name><name><surname>Sachs</surname> <given-names>A</given-names></name><name><surname>Lusis</surname> <given-names>AJ</given-names></name></person-group><year iso-8601-date="2005">2005</year><article-title>An integrative genomics approach to infer causal associations between gene expression and disease</article-title><source>Nature Genetics</source><volume>37</volume><fpage>710</fpage><lpage>717</lpage><pub-id pub-id-type="doi">10.1038/ng1589</pub-id><pub-id pub-id-type="pmid">15965475</pub-id></element-citation></ref><ref id="bib69"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Sharon</surname> <given-names>E</given-names></name><name><surname>Kalma</surname> <given-names>Y</given-names></name><name><surname>Sharp</surname> <given-names>A</given-names></name><name><surname>Raveh-Sadka</surname> <given-names>T</given-names></name><name><surname>Levo</surname> <given-names>M</given-names></name><name><surname>Zeevi</surname> <given-names>D</given-names></name><name><surname>Keren</surname> <given-names>L</given-names></name><name><surname>Yakhini</surname> <given-names>Z</given-names></name><name><surname>Weinberger</surname> <given-names>A</given-names></name><name><surname>Segal</surname> <given-names>E</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>Inferring gene regulatory logic from high-throughput measurements of thousands of systematically designed promoters</article-title><source>Nature Biotechnology</source><volume>30</volume><fpage>521</fpage><lpage>530</lpage><pub-id pub-id-type="doi">10.1038/nbt.2205</pub-id><pub-id pub-id-type="pmid">22609971</pub-id></element-citation></ref><ref id="bib70"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Shively</surname> <given-names>CA</given-names></name><name><surname>Liu</surname> <given-names>J</given-names></name><name><surname>Chen</surname> <given-names>X</given-names></name><name><surname>Loell</surname> <given-names>K</given-names></name><name><surname>Mitra</surname> <given-names>RD</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Homotypic cooperativity and collective binding are determinants of bHLH specificity and function</article-title><source>PNAS</source><volume>116</volume><fpage>16143</fpage><lpage>16152</lpage><pub-id pub-id-type="doi">10.1073/pnas.1818015116</pub-id><pub-id pub-id-type="pmid">31341088</pub-id></element-citation></ref><ref id="bib71"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Shiwa</surname> <given-names>Y</given-names></name><name><surname>Fukushima-Tanaka</surname> <given-names>S</given-names></name><name><surname>Kasahara</surname> <given-names>K</given-names></name><name><surname>Horiuchi</surname> <given-names>T</given-names></name><name><surname>Yoshikawa</surname> <given-names>H</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>Whole-Genome profiling of a novel mutagenesis technique using Proofreading-Deficient DNA polymerase δ</article-title><source>International Journal of Evolutionary Biology</source><volume>2012</volume><elocation-id>860797</elocation-id><pub-id pub-id-type="doi">10.1155/2012/860797</pub-id><pub-id pub-id-type="pmid">22675654</pub-id></element-citation></ref><ref id="bib72"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Sinnott-Armstrong</surname> <given-names>N</given-names></name><name><surname>Naqvi</surname> <given-names>S</given-names></name><name><surname>Rivas</surname> <given-names>M</given-names></name><name><surname>Pritchard</surname> <given-names>JK</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>GWAS of three molecular traits highlights core genes and pathways alongside a highly polygenic background</article-title><source>eLife</source><volume>10</volume><elocation-id>e58615</elocation-id><pub-id pub-id-type="doi">10.7554/eLife.58615</pub-id><pub-id pub-id-type="pmid">33587031</pub-id></element-citation></ref><ref id="bib73"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Stuckey</surname> <given-names>S</given-names></name><name><surname>Mukherjee</surname> <given-names>K</given-names></name><name><surname>Storici</surname> <given-names>F</given-names></name></person-group><year iso-8601-date="2011">2011</year><article-title>In vivo site-specific mutagenesis and gene collage using the delitto perfetto system in yeast <italic>Saccharomyces cerevisiae</italic></article-title><source>Methods in Molecular Biology</source><volume>745</volume><fpage>173</fpage><lpage>191</lpage><pub-id pub-id-type="doi">10.1007/978-1-61779-129-1_11</pub-id><pub-id pub-id-type="pmid">21660695</pub-id></element-citation></ref><ref id="bib74"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Tang</surname> <given-names>Y</given-names></name><name><surname>Gao</surname> <given-names>XD</given-names></name><name><surname>Wang</surname> <given-names>Y</given-names></name><name><surname>Yuan</surname> <given-names>BF</given-names></name><name><surname>Feng</surname> <given-names>YQ</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>Widespread existence of cytosine methylation in yeast DNA measured by gas chromatography/mass spectrometry</article-title><source>Analytical Chemistry</source><volume>84</volume><fpage>7249</fpage><lpage>7255</lpage><pub-id pub-id-type="doi">10.1021/ac301727c</pub-id><pub-id pub-id-type="pmid">22852529</pub-id></element-citation></ref><ref id="bib75"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Tarassov</surname> <given-names>K</given-names></name><name><surname>Messier</surname> <given-names>V</given-names></name><name><surname>Landry</surname> <given-names>CR</given-names></name><name><surname>Radinovic</surname> <given-names>S</given-names></name><name><surname>Serna Molina</surname> <given-names>MM</given-names></name><name><surname>Shames</surname> <given-names>I</given-names></name><name><surname>Malitskaya</surname> <given-names>Y</given-names></name><name><surname>Vogel</surname> <given-names>J</given-names></name><name><surname>Bussey</surname> <given-names>H</given-names></name><name><surname>Michnick</surname> <given-names>SW</given-names></name></person-group><year iso-8601-date="2008">2008</year><article-title>An in vivo map of the yeast protein interactome</article-title><source>Science</source><volume>320</volume><fpage>1465</fpage><lpage>1470</lpage><pub-id pub-id-type="doi">10.1126/science.1153878</pub-id><pub-id pub-id-type="pmid">18467557</pub-id></element-citation></ref><ref id="bib76"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Teixeira</surname> <given-names>MC</given-names></name><name><surname>Monteiro</surname> <given-names>PT</given-names></name><name><surname>Palma</surname> <given-names>M</given-names></name><name><surname>Costa</surname> <given-names>C</given-names></name><name><surname>Godinho</surname> <given-names>CP</given-names></name><name><surname>Pais</surname> <given-names>P</given-names></name><name><surname>Cavalheiro</surname> <given-names>M</given-names></name><name><surname>Antunes</surname> <given-names>M</given-names></name><name><surname>Lemos</surname> <given-names>A</given-names></name><name><surname>Pedreira</surname> <given-names>T</given-names></name><name><surname>Sá-Correia</surname> <given-names>I</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>YEASTRACT: an upgraded database for the analysis of transcription regulatory networks in <italic>Saccharomyces cerevisiae</italic></article-title><source>Nucleic Acids Research</source><volume>46</volume><fpage>D348</fpage><lpage>D353</lpage><pub-id pub-id-type="doi">10.1093/nar/gkx842</pub-id></element-citation></ref><ref id="bib77"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Tirosh</surname> <given-names>I</given-names></name><name><surname>Barkai</surname> <given-names>N</given-names></name></person-group><year iso-8601-date="2008">2008</year><article-title>Two strategies for gene regulation by promoter nucleosomes</article-title><source>Genome Research</source><volume>18</volume><fpage>1084</fpage><lpage>1091</lpage><pub-id pub-id-type="doi">10.1101/gr.076059.108</pub-id></element-citation></ref><ref id="bib78"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Uemura</surname> <given-names>H</given-names></name><name><surname>Koshio</surname> <given-names>M</given-names></name><name><surname>Inoue</surname> <given-names>Y</given-names></name><name><surname>Lopez</surname> <given-names>MC</given-names></name><name><surname>Baker</surname> <given-names>HV</given-names></name></person-group><year iso-8601-date="1997">1997</year><article-title>The role of Gcr1p in the transcriptional activation of glycolytic genes in yeast <italic>Saccharomyces cerevisiae</italic></article-title><source>Genetics</source><volume>147</volume><fpage>521</fpage><lpage>532</lpage><pub-id pub-id-type="doi">10.1093/genetics/147.2.521</pub-id><pub-id pub-id-type="pmid">9335590</pub-id></element-citation></ref><ref id="bib79"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Uphoff</surname> <given-names>S</given-names></name><name><surname>Lord</surname> <given-names>ND</given-names></name><name><surname>Okumus</surname> <given-names>B</given-names></name><name><surname>Potvin-Trottier</surname> <given-names>L</given-names></name><name><surname>Sherratt</surname> <given-names>DJ</given-names></name><name><surname>Paulsson</surname> <given-names>J</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Stochastic activation of a DNA damage response causes cell-to-cell mutation rate variation</article-title><source>Science</source><volume>351</volume><fpage>1094</fpage><lpage>1097</lpage><pub-id pub-id-type="doi">10.1126/science.aac9786</pub-id><pub-id pub-id-type="pmid">26941321</pub-id></element-citation></ref><ref id="bib80"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>van Leeuwen</surname> <given-names>J</given-names></name><name><surname>Pons</surname> <given-names>C</given-names></name><name><surname>Mellor</surname> <given-names>JC</given-names></name><name><surname>Yamaguchi</surname> <given-names>TN</given-names></name><name><surname>Friesen</surname> <given-names>H</given-names></name><name><surname>Koschwanez</surname> <given-names>J</given-names></name><name><surname>Ušaj</surname> <given-names>MM</given-names></name><name><surname>Pechlaner</surname> <given-names>M</given-names></name><name><surname>Takar</surname> <given-names>M</given-names></name><name><surname>Ušaj</surname> <given-names>M</given-names></name><name><surname>VanderSluis</surname> <given-names>B</given-names></name><name><surname>Andrusiak</surname> <given-names>K</given-names></name><name><surname>Bansal</surname> <given-names>P</given-names></name><name><surname>Baryshnikova</surname> <given-names>A</given-names></name><name><surname>Boone</surname> <given-names>CE</given-names></name><name><surname>Cao</surname> <given-names>J</given-names></name><name><surname>Cote</surname> <given-names>A</given-names></name><name><surname>Gebbia</surname> <given-names>M</given-names></name><name><surname>Horecka</surname> <given-names>G</given-names></name><name><surname>Horecka</surname> <given-names>I</given-names></name><name><surname>Kuzmin</surname> <given-names>E</given-names></name><name><surname>Legro</surname> <given-names>N</given-names></name><name><surname>Liang</surname> <given-names>W</given-names></name><name><surname>van Lieshout</surname> <given-names>N</given-names></name><name><surname>McNee</surname> <given-names>M</given-names></name><name><surname>San Luis</surname> <given-names>BJ</given-names></name><name><surname>Shaeri</surname> <given-names>F</given-names></name><name><surname>Shuteriqi</surname> <given-names>E</given-names></name><name><surname>Sun</surname> <given-names>S</given-names></name><name><surname>Yang</surname> <given-names>L</given-names></name><name><surname>Youn</surname> <given-names>JY</given-names></name><name><surname>Yuen</surname> <given-names>M</given-names></name><name><surname>Costanzo</surname> <given-names>M</given-names></name><name><surname>Gingras</surname> <given-names>AC</given-names></name><name><surname>Aloy</surname> <given-names>P</given-names></name><name><surname>Oostenbrink</surname> <given-names>C</given-names></name><name><surname>Murray</surname> <given-names>A</given-names></name><name><surname>Graham</surname> <given-names>TR</given-names></name><name><surname>Myers</surname> <given-names>CL</given-names></name><name><surname>Andrews</surname> <given-names>BJ</given-names></name><name><surname>Roth</surname> <given-names>FP</given-names></name><name><surname>Boone</surname> <given-names>C</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Exploring genetic suppression interactions on a global scale</article-title><source>Science</source><volume>354</volume><elocation-id>aag0839</elocation-id><pub-id pub-id-type="doi">10.1126/science.aag0839</pub-id><pub-id pub-id-type="pmid">27811238</pub-id></element-citation></ref><ref id="bib81"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Vitkup</surname> <given-names>D</given-names></name><name><surname>Sander</surname> <given-names>C</given-names></name><name><surname>Church</surname> <given-names>GM</given-names></name></person-group><year iso-8601-date="2003">2003</year><article-title>The amino-acid mutational spectrum of human genetic disease</article-title><source>Genome Biology</source><volume>4</volume><elocation-id>R72</elocation-id></element-citation></ref><ref id="bib82"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Xiao</surname> <given-names>R</given-names></name><name><surname>Boehnke</surname> <given-names>M</given-names></name></person-group><year iso-8601-date="2009">2009</year><article-title>Quantifying and correcting for the winner's curse in genetic association studies</article-title><source>Genetic Epidemiology</source><volume>33</volume><fpage>453</fpage><lpage>462</lpage><pub-id pub-id-type="doi">10.1002/gepi.20398</pub-id><pub-id pub-id-type="pmid">19140131</pub-id></element-citation></ref><ref id="bib83"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Yagi</surname> <given-names>S</given-names></name><name><surname>Yagi</surname> <given-names>K</given-names></name><name><surname>Fukuoka</surname> <given-names>J</given-names></name><name><surname>Suzuki</surname> <given-names>M</given-names></name></person-group><year iso-8601-date="1994">1994</year><article-title>The UAS of the yeast GAPDH promoter consists of multiple general functional elements including RAP1 and GRF2 binding sites</article-title><source>Journal of Veterinary Medical Science</source><volume>56</volume><fpage>235</fpage><lpage>244</lpage><pub-id pub-id-type="doi">10.1292/jvms.56.235</pub-id></element-citation></ref><ref id="bib84"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Yampolsky</surname> <given-names>LY</given-names></name><name><surname>Stoltzfus</surname> <given-names>A</given-names></name></person-group><year iso-8601-date="2005">2005</year><article-title>The exchangeability of amino acids in proteins</article-title><source>Genetics</source><volume>170</volume><fpage>1459</fpage><lpage>1472</lpage><pub-id pub-id-type="doi">10.1534/genetics.104.039107</pub-id><pub-id pub-id-type="pmid">15944362</pub-id></element-citation></ref><ref id="bib85"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Yao</surname> <given-names>C</given-names></name><name><surname>Joehanes</surname> <given-names>R</given-names></name><name><surname>Johnson</surname> <given-names>AD</given-names></name><name><surname>Huan</surname> <given-names>T</given-names></name><name><surname>Liu</surname> <given-names>C</given-names></name><name><surname>Freedman</surname> <given-names>JE</given-names></name><name><surname>Munson</surname> <given-names>PJ</given-names></name><name><surname>Hill</surname> <given-names>DE</given-names></name><name><surname>Vidal</surname> <given-names>M</given-names></name><name><surname>Levy</surname> <given-names>D</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>Dynamic role of <italic>trans</italic> regulation of gene expression in relation to complex traits</article-title><source>The American Journal of Human Genetics</source><volume>100</volume><fpage>571</fpage><lpage>580</lpage><pub-id pub-id-type="doi">10.1016/j.ajhg.2017.02.003</pub-id><pub-id pub-id-type="pmid">28285768</pub-id></element-citation></ref><ref id="bib86"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Yvert</surname> <given-names>G</given-names></name><name><surname>Brem</surname> <given-names>RB</given-names></name><name><surname>Whittle</surname> <given-names>J</given-names></name><name><surname>Akey</surname> <given-names>JM</given-names></name><name><surname>Foss</surname> <given-names>E</given-names></name><name><surname>Smith</surname> <given-names>EN</given-names></name><name><surname>Mackelprang</surname> <given-names>R</given-names></name><name><surname>Kruglyak</surname> <given-names>L</given-names></name></person-group><year iso-8601-date="2003">2003</year><article-title><italic>Trans</italic>-acting regulatory variation in <italic>Saccharomyces cerevisiae</italic> and the role of transcription factors</article-title><source>Nature Genetics</source><volume>35</volume><fpage>57</fpage><lpage>64</lpage><pub-id pub-id-type="doi">10.1038/ng1222</pub-id><pub-id pub-id-type="pmid">12897782</pub-id></element-citation></ref><ref id="bib87"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Zheng</surname> <given-names>W</given-names></name><name><surname>Zhao</surname> <given-names>H</given-names></name><name><surname>Mancera</surname> <given-names>E</given-names></name><name><surname>Steinmetz</surname> <given-names>LM</given-names></name><name><surname>Snyder</surname> <given-names>M</given-names></name></person-group><year iso-8601-date="2010">2010</year><article-title>Genetic analysis of variation in transcription factor binding in yeast</article-title><source>Nature</source><volume>464</volume><fpage>1187</fpage><lpage>1191</lpage><pub-id pub-id-type="doi">10.1038/nature08934</pub-id><pub-id pub-id-type="pmid">20237471</pub-id></element-citation></ref><ref id="bib88"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Zhu</surname> <given-names>C</given-names></name><name><surname>Byers</surname> <given-names>KJ</given-names></name><name><surname>McCord</surname> <given-names>RP</given-names></name><name><surname>Shi</surname> <given-names>Z</given-names></name><name><surname>Berger</surname> <given-names>MF</given-names></name><name><surname>Newburger</surname> <given-names>DE</given-names></name><name><surname>Saulrieta</surname> <given-names>K</given-names></name><name><surname>Smith</surname> <given-names>Z</given-names></name><name><surname>Shah</surname> <given-names>MV</given-names></name><name><surname>Radhakrishnan</surname> <given-names>M</given-names></name><name><surname>Philippakis</surname> <given-names>AA</given-names></name><name><surname>Hu</surname> <given-names>Y</given-names></name><name><surname>De Masi</surname> <given-names>F</given-names></name><name><surname>Pacek</surname> <given-names>M</given-names></name><name><surname>Rolfs</surname> <given-names>A</given-names></name><name><surname>Murthy</surname> <given-names>T</given-names></name><name><surname>Labaer</surname> <given-names>J</given-names></name><name><surname>Bulyk</surname> <given-names>ML</given-names></name></person-group><year iso-8601-date="2009">2009</year><article-title>High-resolution DNA-binding specificity analysis of yeast transcription factors</article-title><source>Genome Research</source><volume>19</volume><fpage>556</fpage><lpage>566</lpage><pub-id pub-id-type="doi">10.1101/gr.090233.108</pub-id><pub-id pub-id-type="pmid">19158363</pub-id></element-citation></ref></ref-list></back><sub-article article-type="decision-letter" id="sa1"><front-stub><article-id pub-id-type="doi">10.7554/eLife.67806.sa1</article-id><title-group><article-title>Decision letter</article-title></title-group><contrib-group><contrib contrib-type="editor"><name><surname>Landry</surname><given-names>Christian R</given-names></name><role>Reviewing Editor</role><aff><institution>Université Laval</institution><country>Canada</country></aff></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name><surname>He</surname><given-names>Fei</given-names> </name><role>Reviewer</role><aff><institution/></aff></contrib><contrib contrib-type="reviewer"><name><surname>Verta</surname><given-names>Jukka-Pekka</given-names> </name><role>Reviewer</role><aff><institution>University of Helsinki</institution><country>Finland</country></aff></contrib></contrib-group></front-stub><body><boxed-text><p>Our editorial process produces two outputs: i) <ext-link ext-link-type="uri" xlink:href="https://sciety.org/articles/activity/10.1101/2021.02.22.432283">public reviews</ext-link> designed to be posted alongside <ext-link ext-link-type="uri" xlink:href="https://www.biorxiv.org/content/10.1101/2021.02.22.432283v1.full">the preprint</ext-link> for the benefit of readers; ii) feedback on the manuscript for the authors, including requests for revisions, shown below. We also include an acceptance summary that explains what the editors found interesting or important about the work.</p></boxed-text><p><bold>Acceptance summary:</bold></p><p>The relationship between traits and mutations influences the rate and direction in which traits evolve. One key question in evolutionary biology is therefore how traits can be affected by spontaneous mutations. Here, the authors map a set of mutations that affect the expression of a focal gene in yeast, and examine their individual effects and location in the genome and in the regulatory network. The work is rigorous and the results are well presented. The findings will be of great interest for geneticist and evolutionary biologists interested in the evolution of gene expression and of complex traits.</p><p><bold>Decision letter after peer review:</bold></p><p>Thank you for submitting your article &quot;Mutational sources of trans-regulatory variation affecting gene expression in <italic>Saccharomyces cerevisiae</italic>&quot; for consideration by <italic>eLife</italic>. Your article has been reviewed by 3 peer reviewers, and the evaluation has been overseen by a Reviewing Editor and Molly Przeworski as the Senior Editor. The following individuals involved in review of your submission have agreed to reveal their identity: Fei He (Reviewer #2); Jukka-Pekka Verta (Reviewer #3).</p><p>The reviewers have discussed their reviews with one another, and the Reviewing Editor has drafted this to help you prepare a revised submission.</p><p>Essential Revisions:</p><p>The three reviewers and I have appreciated the manuscript in terms of the questions addressed and of the quality of the work. However, we have identified several issues that would need to be addressed. The main points are:</p><p>1) One reviewer comments on the fact that the overlap with previous eQTLs is modest and not completely persuasive, for instance because both approaches could have similar biases and thus produce an enrichment. It would be important to clarify this point. Also, other comments relate to the fact that the overlap with known regulators and previous eQTLs was expected. It would be useful in the introduction to layout the reasons why natural variation (analyzed by eQTL mapping for instance) and more classical functional genetics studies, performed with gene deletions and mutants, would or would not overlap with spontaneous mutants such as the ones analyzed here. For instance, non-sense variants acting in trans can be recovered in experiments like yours but could be rare in nature because of selection acting directly on the genes or on pleiotropic effects. Many trans-regulatory relationships could therefore be missed. Such a justification (based on technical, biological and evolutionary considerations) regarding the need for this type of experiment would also broaden the scope of the paper.</p><p>2) One reviewer mentions that the inability to map some of the causal mutations may be due to the complex and polygenic architecture of such traits. It would be important to address this question in more details. The likelihood of detection of some types of variation may have a strong influence on the conclusions reached so it is important to eliminate or at least take into account these biases. The issue of potential interactions among mutations is also raised by the other reviewers.</p><p>3) As pointed out by reviewer 1, some of the result sections are very descriptive (sections on the number of mutations, their effects, etc.) and sometimes difficult to follow. A more summarized section with visual support could help.</p><p>4) Reviewer 2 raises potentially important questions about the statistical analyses and GO analysis. It would be important to answer those and make the appropriate changes when needed.</p><p>5) Reviewer 3 (and indirectly reviewers 1 and 2) raises questions regarding the generality of the findings for other organisms with different layers of regulation of gene expression. It would be important to address this point for the board readership of <italic>eLife</italic>.</p><p><italic>Reviewer #1 (Recommendations for the authors):</italic></p><p>I recommend greater consideration of the effect-size distribution of trans-acting mutations and acknowledgment that the mutations discovered here- mostly missense and nonsense mutations – may not be typical of the underlying spectrum of trait-relevant mutations. I thought in particular as I was reading of the very recent <italic>eLife</italic> paper from Sinnott-Armstrong et al., showing that molecular traits in humans are hugely polygenic, and that the individually-detectable genes are the ones you'd expect. I don't believe that discussing this viewpoint diminishes the authors' findings at all. It simply requires changing claims like coding mutations are &quot;more likely to affect&quot; to &quot;more likely to be detected&quot; or something like that.</p><p>The mutations discovered by candidate-gene resequencing obviously have a totally different ascertainment model than the ones discovered by mapping, and all analyses should be repeated without those (I think most are already).</p><p>I got quite lost in trying to track the different numbers of strains and mutations through the manuscript. Some kind of table or flowchart would really help. With some effort, I get that there are 82 mutants and 46 of these have BSA hits. Of these 46, 29 have one hit and so we can carry these 29 forward. 13 have two hits, and of these 5 strains give 2 unlinked hits, so that brings 10 more mutations forward. Of the remaining 12 strains with multiple linked mutations, 9 have different G statistics so that brings 9 forward. Now we're at 29+10+9 = 48. It's unclear how we end up with 12 linked-case mutations at line 293. I guess it's nine where there were differences in G statistics and 3 from the functional tests? The source of these last three is not explained very clearly. That gives me 51 mapped mutations (vs 52 at line 291) + 17 candidate gene mutations; I'm not sure how that turns into 66 total at line 291. Without belaboring the point, it's hard to keep track here. Similarly, I don't know where the 1766 non-trans mutations come from. There are 1819 total mutations and 69 causal mutations, leaving 1750 rather than 1766. And finally, I am confused by the report of 23.9 mutations per strain, which does not match 1819 mutations / 82 strains = 22.18.</p><p><italic>Reviewer #2 (Recommendations for the authors):</italic></p><p>In summary, this work is interesting and there is some weakness. It should be done with major revision as my opinion. As always there are a number of unclarities in the manuscript, I would like the authors to elucidate.</p><p>1) As the trans regulation could be polygenic selection, and this should be discussed in the paper, particularly there are 36 mutations with the small effects.</p><p>2) The authors tried to correlate the trans-regulatory mutations are enriched natural variation affecting TDH3 expression. However, the proof of finding is not so strong.</p><p>3) As the authors did a lot of test for the significance, I only found few cases of p value adjustment of post-hoc test.</p><p>4) RNAseq of realtime-PCR can test the expression change of trans regulators because of mutations. if this data is supplied, it will reveal trans changes of function or abundance effects on expression of targets.</p><p>5) The logical structure of the main text can be improved a bit.</p><p>6) In Figure 1B-C, the distribution of relative fluorescence level seems unequal distribution. Is it normal or biased?</p><p>7) In Figure 6B, how and why the authors selected these GO terms for presentation from 156 enriched GO terms in suppl. Data 8? Did the author do cluster for the GO terms? The GO of metabolism almost means everything, and I think it is not meaningful for biological questions.</p><p>8) I am not sure whether aneuploid can classify into trans regulation.</p><p><italic>Reviewer #3 (Recommendations for the authors):</italic></p><p>Line 34. Please be more specific about the role of trans-variation in &quot;with expression variation often derived from trans-regulatory mutations within species.&quot;</p><p>Line 44. Please expand the thought in this last sentence to include implications, instead of simply stating that overlap exists.</p><p>Line 68. There exists studies analysing the mutational target size of trans-effects, it would be appropriate to cite them here.</p><p>Lines 134-136, 138. Please clarify the selection criteria used here – selected at random or are these all the available mutants (based on what criteria)?</p><p>Lines 290-295. By this point, you have referred to multiple sets of SNPs in the different analyses (chosen at random/by targeted analyses, showing association in BSA or not, tested for functional effects or not etc.) and at least I could not follow how the analyses resulted in the summarizing numbers of SNPs presented in this paragraph. I don't have a perfect suggestion to address this problem, but I think a schematic figure would go a long way keep the reader on track.</p><p>Lines 290-291. Is it possible to put these figures into perspective of the whole mutagenesis experiment? What is the total number of mutations created in the experiment, how many of these influence Ptdh3-YFP and how many further have trans effects?</p><p>Line 297. What's the criteria to select exactly these 1766 mutations?</p><p>Lines 685-698. The details explaining the choice of the statistical test would better fit a supplementary note than the methods.</p><p>Lines 515-546 (the whole section). Can you observe any segregating variation in RAP1 or GCR1? An absence of natural genetic variation could indicate that trans effects cannot happen in these genes because of their lethality (case of RAP1), while observable variation could strengthen the case that absence of trans effects is the result of sampling bias (case of GCR1).</p><p>Figure legends (esp. Figure 2 and its supplements). The figure legends are very long and contain e.g. description of methods and results. I would consider condensing the legends as much as possible or alternatively explaining the additional analyses in the supplementary materials. On the other hand I do understand the logic of explaining e.g. supporting analyses in figure supplements and their legends, so I'll leave the choice to the editor and the authors.</p><p>Figure 4. You conclude that &quot;it seems that regulatory networks describing the relationships between transcription factors and target genes might capture only a small fraction of the potential sources of trans-regulatory variation.&quot; – yet, figure 4 describes exactly these kinds of relationships between transcription factors and downstream genes. Would it be possible to illustrate the significance of other trans-acting sources in addition to transcription factors in figure 4?</p></body></sub-article><sub-article article-type="reply" id="sa2"><front-stub><article-id pub-id-type="doi">10.7554/eLife.67806.sa2</article-id><title-group><article-title>Author response</article-title></title-group></front-stub><body><disp-quote content-type="editor-comment"><p>Essential Revisions:</p><p>The three reviewers and I have appreciated the manuscript in terms of the questions addressed and of the quality of the work. However, we have identified several issues that would need to be addressed. The main points are:</p><p>1) One reviewer comments on the fact that the overlap with previous eQTLs is modest and not completely persuasive, for instance because both approaches could have similar biases and thus produce an enrichment. It would be important to clarify this point.</p></disp-quote><p>We respond to this issue more fully below in response to reviewer 1’s more detailed</p><p>comment. In short, we tested whether differences in sequencing depth across the genome, which might be similar in both BSA-seq mapping experiments, could explain the reported overlap between trans-regulatory mutations and eQTLs. We found that variation in sequencing depth was unlikely to explain the observed enrichment of trans-regulatory mutations in regions of the genome previously shown to harbor eQTLs affecting TDH3 expression. A new figure (Figure 7 – supplementary figure 1) has been added to the manuscript showing the results of this analysis.</p><disp-quote content-type="editor-comment"><p>Also, other comments relate to the fact that the overlap with known regulators and previous eQTLs was expected.</p></disp-quote><p>We agree that some overlap was expected (which is why we tested for the enrichment);</p><p>however, the degree to which the trans-regulatory mutants overlapped with known regulators and previous eQTL was not predictable and, to the best of our knowledge, has not previously been tested empirically. For example, we found that only 6% of trans-regulatory mutations mapped to transcription factors previously shown to regulate TDH3; a priori, we would have guessed that this overlap would have been much greater. In fact, we failed to recover mutations in the two best characterized direct regulators of TDH3 (Rap1p and Gcr1p), which we further investigated in the paper. As stated in the paper, this observation suggests that:</p><p>“regulatory networks describing the relationships between transcription factors and target genes might capture only a small fraction of the potential sources of transregulatory variation.”</p><p>Nonetheless, we have tried to better convey the expected overlap between known regulators and the trans-regulatory mutants by modifying the sentence on line 448 to read:</p><p>“Therefore, the inferred regulatory network had predictive power as expected, but the vast majority of trans-regulatory coding mutations (61 of 65, or 94%) mapped to genes outside of this network.”</p><p>As for the overlap with eQTL, we agree that an exhaustive study of mutations affecting</p><p>expression of TDH3 should overlap with eQTL affecting TDH3 expression, but it was less clear that the set of 69 mutations we mapped would be sufficient to see an enrichment. For example, if the mutations we mapped were more likely to be deleterious than the variants responsible for the eQTL mapped between strains (for instance if they tended to have larger or more pleiotropic effects), we might not have seen a significant overlap. We might also not have seen an overlap if epistatic interactions among natural variants were required to impact expression of TDH3 expression because we only looked at the effects of single trans-regulatory mutations in this study. These points have been added to the revised Discussion (which was a Conclusions section in the original submission), as suggested in the rest of the editor’s comment shown below.</p><disp-quote content-type="editor-comment"><p>It would be useful in the introduction to layout the reasons why natural variation (analyzed by eQTL mapping for instance) and more classical functional genetics studies, performed with gene deletions and mutants, would or would not overlap with spontaneous mutants such as the ones analyzed here. For instance, non-sense variants acting in trans can be recovered in experiments like yours but could be rare in nature because of selection acting directly on the genes or on pleiotropic effects. Many trans-regulatory relationships could therefore be missed. Such a justification (based on technical, biological and evolutionary considerations) regarding the need for this type of experiment would also broaden the scope of the paper.</p></disp-quote><p>As requested, we have now added such a discussion to the introduction in a new second</p><p>paragraph (lines 68-80) and by modifying the end of the third paragraph and the beginning of the fourth paragraph (lines 93-105). We have also addressed these topics in the revised discussion (lines 640-662).</p><disp-quote content-type="editor-comment"><p>2) One reviewer mentions that the inability to map some of the causal mutations may be due to the complex and polygenic architecture of such traits. It would be important to address this question in more details.</p></disp-quote><p>While we agree that natural variation in gene expression has a complex and polygenic</p><p>architecture, we mapped mutations in this study from mutant genotypes with only ~24</p><p>mutations spread throughout the entire genome (compared to ~33,000 to ~56,000 SNPs, or ~2.8 to 4.6 SNP/kb, between strains of <italic>S. cerevisiae</italic> used in the eQTL mapping study). That is, we intentionally designed this experiment to try to capture changes in activity of the TDH3 promoter caused by single mutations. Nonetheless, our inability to map some of the causal mutations might have been due to their small individual effects or interactions between mutations, which are possibilities described and explored in paragraph on lines 235-249 and in Figure 2 —figure supplement 2 and 3. We also added the following sentence on line 149 to try to make this aspect of the experimental design more clear:</p><p>“The dose of EMS used in these studies was chosen so that most mutants with a detectable change in PTDH3-YFP expression should have only one mutation causing this change in expression among the mutations they carry (Metzger et al., 2016; Gruber et al., 2012).”</p><disp-quote content-type="editor-comment"><p>The likelihood of detection of some types of variation may have a strong influence on the conclusions reached so it is important to eliminate or at least take into account these biases.</p></disp-quote><p>To address this concern, which we assume to be primarily related to the effect size of</p><p>mutations we could map as expanded upon below by reviewer 1, we have modified the text to limit the scope of our conclusions (e.g., line 383) and included this limitation of the work in the revised discussion (lines 666-669).</p><disp-quote content-type="editor-comment"><p>The issue of potential interactions among mutations is also raised by the other reviewers.</p></disp-quote><p>We have added text to the discussion acknowledging the importance of considering</p><p>epistasis when studying regulatory variation and identifying this as an area ripe for future study (lines 656-662). We think our data have little to say about genetic interactions because most mutant phenotypes we studied were caused by single mutations. Specifically, the individual mapped mutations explained 94% of the variation in expression observed among the original EMS mutants (Figure 2G), each of which carried an additional 20-30 other mutations. These data suggest that for the mutants we examined, epistatic interactions had negligible effects on expression of P<sub>TDH3</sub>-YFP.</p><disp-quote content-type="editor-comment"><p>3) As pointed out by reviewer 1, some of the result sections are very descriptive (sections on the number of mutations, their effects, etc.) and sometimes difficult to follow. A more summarized section with visual support could help.</p></disp-quote><p>We thank reviewers 1 and 3 for suggesting the idea of a flowchart to help tracking the</p><p>number of mutations and mutants at different steps of the study. We have followed their suggestion and added such a diagram as a new Figure 1 —figure supplement 1.</p><disp-quote content-type="editor-comment"><p>4) Reviewer 2 raises potentially important questions about the statistical analyses and GO analysis. It would be important to answer those and make the appropriate changes when needed.</p></disp-quote><p>We respond more fully to these points in response to reviewer 2’s comments below.</p><disp-quote content-type="editor-comment"><p>5) Reviewer 3 (and indirectly reviewers 1 and 2) raises questions regarding the generality of the findings for other organisms with different layers of regulation of gene expression. It would be important to address this point for the board readership of eLife.</p></disp-quote><p>To fully address this comment, we expanded the former Conclusions section into a more</p><p>complete Discussion section and specifically addressed issues related to generalizability in the final paragraph (lines 682-697).</p><disp-quote content-type="editor-comment"><p>Reviewer #1 (Recommendations for the authors):</p><p>I recommend greater consideration of the effect-size distribution of trans-acting mutations and acknowledgment that the mutations discovered here- mostly missense and nonsense mutations – may not be typical of the underlying spectrum of trait-relevant mutations. I thought in particular as I was reading of the very recent eLife paper from Sinnott-Armstrong et al., showing that molecular traits in humans are hugely polygenic, and that the individually-detectable genes are the ones you'd expect. I don't believe that discussing this viewpoint diminishes the authors' findings at all. It simply requires changing claims like coding mutations are &quot;more likely to affect&quot; to &quot;more likely to be detected&quot; or something like that.</p></disp-quote><p>We thank the reviewer for this comment. We have now revised the introduction to make more explicit the differences between mutations identified and characterized in this study and variation segregating in the wild. We have also made changes to qualify some statements, such as that coding mutations are more likely to affect expression by 3% or more (line 383), which limits the scope of the statement to the size of effects we were able to examine. Finally, we have added explicit statements about the limitations of the study to the revised discussion.</p><disp-quote content-type="editor-comment"><p>The mutations discovered by candidate-gene resequencing obviously have a totally different ascertainment model than the ones discovered by mapping, and all analyses should be repeated without those (I think most are already).</p></disp-quote><p>We have now repeated all analyses without the 17 mutations identified by sequencing</p><p>candidate genes. These results are reported in Figure 3 —figure supplement 1, Figure 3 - figure supplement 2, Figure 3 —figure supplement 4, Supplementary File 8 and in multiple parts of the Results section. The same patterns were observed and the significance of statistical tests was not affected by the removal of the 17 mutations, thus our conclusions did not change. In particular, the results of the GO enrichment analysis were very similar because this analysis was focused on genes and not on mutations: only two genes were excluded (ADE2 and ADE6) when we removed the 17 mutations identified by sequencing candidate genes.</p><disp-quote content-type="editor-comment"><p>I got quite lost in trying to track the different numbers of strains and mutations through the manuscript. Some kind of table or flowchart would really help. With some effort, I get that there are 82 mutants and 46 of these have BSA hits. Of these 46, 29 have one hit and so we can carry these 29 forward. 13 have two hits, and of these 5 strains give 2 unlinked hits, so that brings 10 more mutations forward. Of the remaining 12 strains with multiple linked mutations, 9 have different G statistics so that brings 9 forward. Now we're at 29+10+9 = 48. It's unclear how we end up with 12 linked-case mutations at line 293. I guess it's nine where there were differences in G statistics and 3 from the functional tests? The source of these last three is not explained very clearly. That gives me 51 mapped mutations (vs 52 at line 291) + 17 candidate gene mutations; I'm not sure how that turns into 66 total at line 291. Without belaboring the point, it's hard to keep track here. Similarly, I don't know where the 1766 non-trans mutations come from. There are 1819 total mutations and 69 causal mutations, leaving 1750 rather than 1766. And finally, I am confused by the report of 23.9 mutations per strain, which does not match 1819 mutations / 82 strains = 22.18.</p></disp-quote><p>We appreciate the efforts made by the reviewer to point out numbers for which the sources needed to be clarified. These comments helped us design a diagram showing the number of mutations and mutants included at each step of the study (Figure 1 —figure supplement 1). We hope this diagram will help readers to track these numbers throughout the manuscript much more easily.</p><disp-quote content-type="editor-comment"><p>Reviewer #2 (Recommendations for the authors):</p><p>In summary, this work is interesting and there is some weakness. It should be done with major revision as my opinion. As always there are a number of unclarities in the manuscript, I would like the authors to elucidate.</p><p>1) As the trans regulation could be polygenic selection, and this should be discussed in the paper, particularly there are 36 mutations with the small effects.</p></disp-quote><p>We agree that trans-regulatory variation segregating in the wild is generally polygenic and have revised the introduction to include this point more specifically. Indeed, our group has previously shown that trans-regulatory variation affecting TDH3 promoter activity in <italic>S. cerevisiae</italic> is highly polygenic (Metzger et al., 2019 Evolution Letters). However, the goal for this work was to identify individual, trans-regulatory mutations affecting TDH3 promoter activity. For this reason, the genotypes used for mapping trans-regulatory mutations were generated with a low dose of EMS that introduced on average 24 mutations per genome, compared to ~34,000 genetic differences between the BY and M22 strains of <italic>S. cerevisiae</italic> and ~54,000 genetic differences between BY and either SK1 or YPS1000. In most cases, only one of these ~24 mutations fully explained the changes in TDH3 expression seen in the mutant. For the 36 mutants in which we were unable to identify a single causative mutation, small effect sizes are one of the possible explanations, as described more fully in the paragraph on lines 235-249.</p><disp-quote content-type="editor-comment"><p>2) The authors tried to correlate the trans-regulatory mutations are enriched natural variation affecting TDH3 expression. However, the proof of finding is not so strong.</p></disp-quote><p>As described above in response to reviewer 1, we have included additional analyses to test for possible biases associated with BSA-seq in the two studies that might contribute to the observed statistically significant overlap in the revised manuscript. We are also not sure what the reviewer means by “not so strong” in this context; the overlap reported was supported statistically by a p-value of P = 9.6 x 10-<sup>5</sup>.</p><disp-quote content-type="editor-comment"><p>3) As the authors did a lot of test for the significance, I only found few cases of p value adjustment of post-hoc test.</p></disp-quote><p>We are guessing that the reviewer is asking about multiple testing corrections rather than post-hoc tests, as we used a false discovery rate correction for multiple tests in Figure 2-supplement 5A. Although we did not use a multiple test correction for the BSA-seq data, we used a conservative significance threshold of 0.001 that was expected to result in a 3.5% false positive rate. Perhaps more importantly, we functionally validated the effects of 40 of the 41 associated mutations tested.</p><disp-quote content-type="editor-comment"><p>4) RNAseq of realtime-PCR can test the expression change of trans regulators because of mutations. if this data is supplied, it will reveal trans changes of function or abundance effects on expression of targets.</p></disp-quote><p>We agree that it would be interesting to know whether the mutations mapped also affected expression of the trans-regulators, but we do not think the additional experiments that would be required to address this point are necessary to support the conclusions presented.</p><disp-quote content-type="editor-comment"><p>5) The logical structure of the main text can be improved a bit.</p></disp-quote><p>We hope that the changes made to the revised manuscript, including a new flow chart added as Figure 1 —figure supplement 1, will help readers follow the work more easily.</p><disp-quote content-type="editor-comment"><p>6) In Figure 1B-C, the distribution of relative fluorescence level seems unequal distribution. Is it normal or biased?</p></disp-quote><p>Distributions of mutational effects for trans-regulatory mutations impacting gene expression have previously been described for the TDH3 promoter (Metzger et al., 2016) as well as for promoters from other genes (Hodgins-Davis et al., 2019). None of these distributions of mutational effects are consistent with a normal distribution. The distribution of mutational effects on P<sub>TDH3</sub>-YFP expression was found to have lower kurtosis (more mutations with large effects) than a normal distribution but was symmetrical (no significant skew, which is what we think the reviewer means by bias).</p><disp-quote content-type="editor-comment"><p>7) In Figure 6B, how and why the authors selected these GO terms for presentation from 156 enriched GO terms in suppl. Data 8? Did the author do cluster for the GO terms? The GO of metabolism almost means everything, and I think it is not meaningful for biological questions.</p></disp-quote><p>We did not represent all 156 enriched GO terms on Figure 6B both to make the figure easier to read and because many GO terms were redundant due to the hierarchical structure of GO terms, as described above in response to this reviewer’s public comments. Instead, we only represented the most specific GO terms corresponding to the end tips of the GO hierarchy. We then further grouped enriched GO terms into four larger categories that are not formal GO terms to provide a broader overview of the shared properties of trans-regulatory mutations (all statistical analyses were performed with the formal GO terms, not on the four broader categories).</p><p>We think metabolism is a meaningful category because (1) it has been widely defined in the literature, (2) it is a commonly used term, and (3) some, but not all, genes in this group encode proteins involved in metabolic pathways. We also think it makes a meaningful distinction between, for example, genes encoding enzymes acting in the de novo purine biosynthesis pathway (ADE2, ADE4, ADE5 and ADE6), which are clearly involved in metabolism, and genes encoding proteins involved in other biological processes such as nucleosome positioning (TUP1 and CHD1), which are not always involved in metabolism.</p><disp-quote content-type="editor-comment"><p>8) I am not sure whether aneuploid can classify into trans regulation.</p></disp-quote><p>A trans-regulatory mutation is defined as any mutation that impacts expression of both</p><p>alleles of a gene in diploid cells. Such effects are generally mediated by diffusible molecules such as RNAs or proteins. Aneuploidies are thus classified as trans-acting mutations when they impact expression of genes located on other chromosomes (as is the case in this study) because their impacts are presumably caused by changes in the abundance of trans-acting proteins and/or RNAs produced from genes on the aneuploid chromosome.</p><disp-quote content-type="editor-comment"><p>Reviewer #3 (Recommendations for the authors):</p><p>Line 34. Please be more specific about the role of trans-variation in &quot;with expression variation often derived from trans-regulatory mutations within species.&quot;</p></disp-quote><p>This part of the sentence has been removed in the revised abstract to avoid potential</p><p>confusion.</p><disp-quote content-type="editor-comment"><p>Line 44. Please expand the thought in this last sentence to include implications, instead of simply stating that overlap exists.</p></disp-quote><p>This sentence has been completely re-written in the revised abstract. We think that the</p><p>revised version addresses this comment as well as the previous comment while also helping to clarify the relationship between trans-regulatory mutations and trans-regulatory variation segregating within a species.</p><disp-quote content-type="editor-comment"><p>Line 68. There exists studies analysing the mutational target size of trans-effects, it would be appropriate to cite them here.</p></disp-quote><p>Empirical data measuring the target size for trans-regulatory mutations is very limited (e.g., Metzger et al., 2016); however, the expectation for a large mutational target size for transregulatory mutations has been described by many authors. This expectation is derived from the fact that each gene’s expression is influenced by proteins and RNAs encoded by many other genes and is described most fully in Hill et al., (2021). A citation to this review has been added to this sentence.</p><disp-quote content-type="editor-comment"><p>Lines 134-136, 138. Please clarify the selection criteria used here – selected at random or are these all the available mutants (based on what criteria)?</p></disp-quote><p>We added the following sentence on line 172 to clarify how the mutants were selected:</p><p>“Overall, the 82 mutants were selected randomly from the 528 EMS mutants that showed statistically significant fluorescence changes greater than 1% relative to wild-type (P &lt; 0.05, see Methods and Figure 1 legend for a description of the statistical tests).”</p><disp-quote content-type="editor-comment"><p>Lines 290-295. By this point, you have referred to multiple sets of SNPs in the different analyses (chosen at random/by targeted analyses, showing association in BSA or not, tested for functional effects or not etc.) and at least I could not follow how the analyses resulted in the summarizing numbers of SNPs presented in this paragraph. I don't have a perfect suggestion to address this problem, but I think a schematic figure would go a long way keep the reader on track.</p></disp-quote><p>The revised version of the manuscript includes a new diagram (Figure 1 —figure supplement 1) that shows how many mutations and mutant strains were included at each step of the study.</p><disp-quote content-type="editor-comment"><p>Lines 290-291. Is it possible to put these figures into perspective of the whole mutagenesis experiment? What is the total number of mutations created in the experiment, how many of these influence Ptdh3-YFP and how many further have trans effects?</p></disp-quote><p>We hope that the new diagram Figure 1 —figure supplement 1 clarifies how the numbers mentioned in the original lines 290-291 were obtained from the mutagenesis experiments. The questions posed about the whole mutagenesis experiment are addressed in the Gruber et al. 2012 and Metzger et al., 2016 papers from which the mutants analyzed here were described. Both of these papers attempt to answer these questions, but the questions are harder to answer than they might seem and require making some assumptions about the data. Answering these questions definitively would require completing the sequencing and mapping described here for 82 mutants for all nearly 2000 EMS mutants analyzed in those studies.</p><disp-quote content-type="editor-comment"><p>Line 297. What's the criteria to select exactly these 1766 mutations?</p></disp-quote><p>We added the following sentences at the end of the paragraph on line 336 to clarify how the 1766 non-regulatory mutations were selected:</p><p>“To identify trends in the properties of these 69 trans-regulatory mutations, we compared them to 1766 mutations considered non-regulatory regarding PTDH3-YFP expression because they showed no significant association with expression of the reporter gene in the BSA-Seq experiment (G-test: P &gt; 0.01, Figure 1 —figure supplement 1). To be more conservative, 8 mutations that showed a marginally significant association with expression (G-test: 0.001 &lt; P &lt; 0.01) as well as 15 mutations associated with expression only because of genetic linkage were excluded from further analyses.”</p><disp-quote content-type="editor-comment"><p>Lines 685-698. The details explaining the choice of the statistical test would better fit a supplementary note than the methods.</p></disp-quote><p>While we agree with the reviewer that these lines are not strictly methodological, we do think they are needed to justify the analysis methods used. We could certainly move them to a supplementary note, but we prefer to keep them here to prevent the reader from needing to flip back and forth between sections and hope that this choice is within the author’s discretion.</p><disp-quote content-type="editor-comment"><p>Lines 515-546 (the whole section). Can you observe any segregating variation in RAP1 or GCR1? An absence of natural genetic variation could indicate that trans effects cannot happen in these genes because of their lethality (case of RAP1), while observable variation could strengthen the case that absence of trans effects is the result of sampling bias (case of GCR1).</p></disp-quote><p>We could search for segregating variation in the coding sequences of RAP1 and GCR1</p><p>using genomic sequences for strains of <italic>S. cerevisiae</italic>, but without experimentally testing their effects it would not be possible to know whether any variants that exist affect activity of the TDH3 promoter. Our data show that several mutations in RAP1 and GCR1 are viable, so finding RAP1 and GCR1 variants segregating in natural populations would not be surprising or easy to interpret.</p><disp-quote content-type="editor-comment"><p>Figure legends (esp. Figure 2 and its supplements). The figure legends are very long and contain e.g. description of methods and results. I would consider condensing the legends as much as possible or alternatively explaining the additional analyses in the supplementary materials. On the other hand I do understand the logic of explaining e.g. supporting analyses in figure supplements and their legends, so I'll leave the choice to the editor and the authors.</p></disp-quote><p>We appreciate the reviewer pointing this out. We have edited the Figure 2 legend to shorten it and remove redundancy with the main text and methods. We also revisited the long Figure 2 —figure supplement 5 legend. In this case, we opted to leave it as is rather than create a separate supplementary text section that refers to this figure. We think that the current format will allow the reader to more easily follow the different hypotheses being tested with the different analyses. Because legends for figure supplements appear online only (i.e., not in the PDF version of the paper), we hope that the extra length there will be less problematic.</p><disp-quote content-type="editor-comment"><p>Figure 4. You conclude that &quot;it seems that regulatory networks describing the relationships between transcription factors and target genes might capture only a small fraction of the potential sources of trans-regulatory variation.&quot; – yet, figure 4 describes exactly these kinds of relationships between transcription factors and downstream genes. Would it be possible to illustrate the significance of other trans-acting sources in addition to transcription factors in figure 4?</p></disp-quote><p>Figure 4 was specifically designed to test whether trans-regulatory mutations tended to lie in transcription factors previously implicated in regulating TDH3 expression, with edges in the network representing direct binding of a transcription factor to a target gene. Because the molecular mechanism by which most other genes harboring trans-regulatory mutations impact expression of TDH3 is not clear, we think it would be challenging to add these other types of genes into the figure without creating confusion. However, to better convey that this figure illustrates only a subset of the trans-regulatory mutations and non-regulatory mutations included in this study, we have modified the figure to show the number of total trans-regulatory and non-regulatory mutations included in the figure.</p></body></sub-article></article>