<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Archiving and Interchange DTD v1.1 20151215//EN"  "JATS-archivearticle1.dtd"><article xmlns:ali="http://www.niso.org/schemas/ali/1.0/" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article" dtd-version="1.1"><front><journal-meta><journal-id journal-id-type="nlm-ta">elife</journal-id><journal-id journal-id-type="publisher-id">eLife</journal-id><journal-title-group><journal-title>eLife</journal-title></journal-title-group><issn publication-format="electronic" pub-type="epub">2050-084X</issn><publisher><publisher-name>eLife Sciences Publications, Ltd</publisher-name></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">62669</article-id><article-id pub-id-type="doi">10.7554/eLife.62669</article-id><article-categories><subj-group subj-group-type="display-channel"><subject>Research Article</subject></subj-group><subj-group subj-group-type="heading"><subject>Genetics and Genomics</subject></subj-group></article-categories><title-group><article-title>Systematic identification of <italic>cis</italic>-regulatory variants that cause gene expression differences in a yeast cross</article-title></title-group><contrib-group><contrib contrib-type="author" id="author-205009" equal-contrib="yes"><name><surname>Renganaath</surname><given-names>Kaushik</given-names></name><contrib-id contrib-id-type="orcid" authenticated="true">https://orcid.org/0000-0003-1010-3604</contrib-id><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="fn" rid="equal-contrib1">†</xref><xref ref-type="fn" rid="con1"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author" id="author-205010" equal-contrib="yes"><name><surname>Chong</surname><given-names>Rockie</given-names></name><contrib-id contrib-id-type="orcid" authenticated="true">https://orcid.org/0000-0001-6736-9687</contrib-id><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="fn" rid="equal-contrib1">†</xref><xref ref-type="fn" rid="con2"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author" id="author-106415"><name><surname>Day</surname><given-names>Laura</given-names></name><xref ref-type="aff" rid="aff3">3</xref><xref ref-type="aff" rid="aff4">4</xref><xref ref-type="aff" rid="aff5">5</xref><xref ref-type="fn" rid="con3"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author" id="author-108438"><name><surname>Kosuri</surname><given-names>Sriram</given-names></name><contrib-id contrib-id-type="orcid" authenticated="true">http://orcid.org/0000-0002-4661-0600</contrib-id><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="other" rid="fund5"/><xref ref-type="other" rid="fund6"/><xref ref-type="other" rid="fund8"/><xref ref-type="fn" rid="con4"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author" corresp="yes" id="author-15412"><name><surname>Kruglyak</surname><given-names>Leonid</given-names></name><contrib-id contrib-id-type="orcid" authenticated="true">https://orcid.org/0000-0002-8065-3057</contrib-id><email>LKruglyak@mednet.ucla.edu</email><xref ref-type="aff" rid="aff3">3</xref><xref ref-type="aff" rid="aff4">4</xref><xref ref-type="aff" rid="aff5">5</xref><xref ref-type="other" rid="fund2"/><xref ref-type="other" rid="fund7"/><xref ref-type="fn" rid="con5"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author" corresp="yes" id="author-74458"><name><surname>Albert</surname><given-names>Frank W</given-names></name><contrib-id contrib-id-type="orcid" authenticated="true">https://orcid.org/0000-0002-1380-8063</contrib-id><email>falbert@umn.edu</email><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="other" rid="fund1"/><xref ref-type="other" rid="fund3"/><xref ref-type="other" rid="fund4"/><xref ref-type="fn" rid="con6"/><xref ref-type="fn" rid="conf1"/></contrib><aff id="aff1"><label>1</label><institution>Department of Genetics, Cell Biology, &amp; Development, University of Minnesota</institution><addr-line><named-content content-type="city">Minneapolis</named-content></addr-line><country>United States</country></aff><aff id="aff2"><label>2</label><institution>Department of Chemistry &amp; Biochemistry, University of California, Los Angeles</institution><addr-line><named-content content-type="city">Los Angeles</named-content></addr-line><country>United States</country></aff><aff id="aff3"><label>3</label><institution>Department of Human Genetics, University of California, Los Angeles</institution><addr-line><named-content content-type="city">Los Angeles</named-content></addr-line><country>United States</country></aff><aff id="aff4"><label>4</label><institution>Department of Biological Chemistry, University of California, Los Angeles</institution><addr-line><named-content content-type="city">Los Angeles</named-content></addr-line><country>United States</country></aff><aff id="aff5"><label>5</label><institution>Howard Hughes Medical Institute, University of California, Los Angeles</institution><addr-line><named-content content-type="city">Los Angeles</named-content></addr-line><country>United States</country></aff></contrib-group><contrib-group content-type="section"><contrib contrib-type="editor"><name><surname>Landry</surname><given-names>Christian R</given-names></name><role>Reviewing Editor</role><aff><institution>Université Laval</institution><country>Canada</country></aff></contrib><contrib contrib-type="senior_editor"><name><surname>Wittkopp</surname><given-names>Patricia J</given-names></name><role>Senior Editor</role><aff><institution>University of Michigan</institution><country>United States</country></aff></contrib></contrib-group><author-notes><fn fn-type="con" id="equal-contrib1"><label>†</label><p>These authors contributed equally to this work</p></fn></author-notes><pub-date date-type="publication" publication-format="electronic"><day>12</day><month>11</month><year>2020</year></pub-date><pub-date pub-type="collection"><year>2020</year></pub-date><volume>9</volume><elocation-id>e62669</elocation-id><history><date date-type="received" iso-8601-date="2020-09-02"><day>02</day><month>09</month><year>2020</year></date><date date-type="accepted" iso-8601-date="2020-11-11"><day>11</day><month>11</month><year>2020</year></date></history><permissions><copyright-statement>© 2020, Renganaath et al</copyright-statement><copyright-year>2020</copyright-year><copyright-holder>Renganaath et al</copyright-holder><ali:free_to_read/><license xlink:href="http://creativecommons.org/licenses/by/4.0/"><ali:license_ref>http://creativecommons.org/licenses/by/4.0/</ali:license_ref><license-p>This article is distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="http://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution License</ext-link>, which permits unrestricted use and redistribution provided that the original author and source are credited.</license-p></license></permissions><self-uri content-type="pdf" xlink:href="elife-62669-v2.pdf"/><abstract><p>Sequence variation in regulatory DNA alters gene expression and shapes genetically complex traits. However, the identification of individual, causal regulatory variants is challenging. Here, we used a massively parallel reporter assay to measure the <italic>cis</italic>-regulatory consequences of 5832 natural DNA variants in the promoters of 2503 genes in the yeast <italic>Saccharomyces cerevisiae</italic>. We identified 451 causal variants, which underlie genetic loci known to affect gene expression. Several promoters harbored multiple causal variants. In five promoters, pairs of variants showed non-additive, epistatic interactions. Causal variants were enriched at conserved nucleotides, tended to have low derived allele frequency, and were depleted from promoters of essential genes, which is consistent with the action of negative selection. Causal variants were also enriched for alterations in transcription factor binding sites. Models integrating these features provided modest, but statistically significant, ability to predict causal variants. This work revealed a complex molecular basis for <italic>cis</italic>-acting regulatory variation.</p></abstract><kwd-group kwd-group-type="author-keywords"><kwd>eQTL</kwd><kwd>genetic variation</kwd><kwd>evolution</kwd><kwd>gene regulation</kwd></kwd-group><kwd-group kwd-group-type="research-organism"><title>Research organism</title><kwd><italic>S. cerevisiae</italic></kwd></kwd-group><funding-group><award-group id="fund1"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100000002</institution-id><institution>National Institutes of Health</institution></institution-wrap></funding-source><award-id>R35GM124676</award-id><principal-award-recipient><name><surname>Albert</surname><given-names>Frank Wolfgang</given-names></name></principal-award-recipient></award-group><award-group id="fund2"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100000011</institution-id><institution>Howard Hughes Medical Institute</institution></institution-wrap></funding-source><principal-award-recipient><name><surname>Kruglyak</surname><given-names>Leonid</given-names></name></principal-award-recipient></award-group><award-group id="fund3"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100000875</institution-id><institution>Pew Charitable Trusts</institution></institution-wrap></funding-source><principal-award-recipient><name><surname>Albert</surname><given-names>Frank Wolfgang</given-names></name></principal-award-recipient></award-group><award-group id="fund4"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100000879</institution-id><institution>Alfred P. Sloan Foundation</institution></institution-wrap></funding-source><principal-award-recipient><name><surname>Albert</surname><given-names>Frank Wolfgang</given-names></name></principal-award-recipient></award-group><award-group id="fund5"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100005665</institution-id><institution>Kinship Foundation</institution></institution-wrap></funding-source><principal-award-recipient><name><surname>Kosuri</surname><given-names>Sriram</given-names></name></principal-award-recipient></award-group><award-group id="fund6"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100004944</institution-id><institution>Department of Energy, Labor and Economic Growth</institution></institution-wrap></funding-source><award-id>DE-FC02-02ER63421</award-id><principal-award-recipient><name><surname>Kosuri</surname><given-names>Sriram</given-names></name></principal-award-recipient></award-group><award-group id="fund7"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100000002</institution-id><institution>National Institutes of Health</institution></institution-wrap></funding-source><award-id>R01GM102308</award-id><principal-award-recipient><name><surname>Kruglyak</surname><given-names>Leonid</given-names></name></principal-award-recipient></award-group><award-group id="fund8"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100000002</institution-id><institution>National Institutes of Health</institution></institution-wrap></funding-source><award-id>DP2GM114829</award-id><principal-award-recipient><name><surname>Kosuri</surname><given-names>Sriram</given-names></name></principal-award-recipient></award-group><funding-statement>The funders had no role in study design, data collection and interpretation, or the decision to submit the work for publication.</funding-statement></funding-group><custom-meta-group><custom-meta specific-use="meta-only"><meta-name>Author impact statement</meta-name><meta-value>Yeast promoters can harbor multiple natural DNA variants that influence gene expression, interact genetically, evolve under negative selection, alter transcription factor motifs, and remain challenging to predict.</meta-value></custom-meta></custom-meta-group></article-meta></front><body><sec sec-type="intro" id="s1"><title>Introduction</title><p>Individual genomes carry thousands of sequence differences in gene-regulatory elements. Collectively, these variants contribute to variation in many phenotypic traits by altering the expression of one or multiple genes (<xref ref-type="bibr" rid="bib6">Albert and Kruglyak, 2015</xref>). The presence of individual DNA variants that alter gene expression can be detected by mapping genomic regions called ‘expression quantitative trait loci’ (eQTLs). Among these, ‘local’ eQTLs are located close to or in the gene whose expression they influence. Eukaryotic species ranging from yeast to human carry large amounts of local regulatory variation (<xref ref-type="bibr" rid="bib12">Brem et al., 2002</xref>; <xref ref-type="bibr" rid="bib39">Hasin-Brumshtein et al., 2016</xref>; <xref ref-type="bibr" rid="bib40">Heyne et al., 2014</xref>; <xref ref-type="bibr" rid="bib88">Rockman et al., 2010</xref>; <xref ref-type="bibr" rid="bib105">Stranger et al., 2005</xref>; <xref ref-type="bibr" rid="bib115">West et al., 2007</xref>). In human populations, most genes are affected by one or multiple local eQTLs (<xref ref-type="bibr" rid="bib36">GTEx Consortium et al., 2017</xref>). Similarly, in a cross between two genetically different yeast isolates, 74% of genes are influenced by local eQTLs (<xref ref-type="bibr" rid="bib4">Albert et al., 2018</xref>). Most of these local eQTLs arise from DNA variants that perturb <italic>cis</italic>-acting regulatory mechanisms (<xref ref-type="bibr" rid="bib4">Albert et al., 2018</xref>; <xref ref-type="bibr" rid="bib89">Ronald et al., 2005</xref>). When such <italic>cis</italic>-acting variants are located in a gene’s transcribed region, they may alter mRNA stability, splicing, polyadenylation, or regulation by RNA-binding proteins. <italic>Cis</italic>-acting variants in promoters or enhancers may alter the transcription of their target genes.</p><p>As a consequence of genetic linkage in experimental crosses (<xref ref-type="bibr" rid="bib4">Albert et al., 2018</xref>) and linkage disequilibrium in outbred populations (<xref ref-type="bibr" rid="bib36">GTEx Consortium et al., 2017</xref>; <xref ref-type="bibr" rid="bib53">Kita et al., 2017</xref>), regions mapped as eQTLs almost always contain multiple sequence variants. Typically, it is assumed that most of these variants have no effect, obscuring the identity of one or a few causal variants in each eQTL (<xref ref-type="fig" rid="fig1">Figure 1</xref>). Although the causal variants in several local eQTLs have been identified (<xref ref-type="bibr" rid="bib17">Chang et al., 2013</xref>; <xref ref-type="bibr" rid="bib21">Claussnitzer et al., 2015</xref>; <xref ref-type="bibr" rid="bib67">Lutz et al., 2019</xref>; <xref ref-type="bibr" rid="bib78">Musunuru et al., 2010</xref>; <xref ref-type="bibr" rid="bib89">Ronald et al., 2005</xref>), most causal variants remain unknown. Because of this lack of systematic information, many questions about local eQTLs remain open, including whether local eQTLs are typically caused by one or multiple variants, if multiple variants interact in a non-additive fashion, what evolutionary forces act on causal variants, which molecular mechanisms causal variants perturb, and whether it may ultimately be possible to combine the answers to these questions to construct models that can predict the consequences of regulatory variants from genome sequence.</p><fig-group><fig id="fig1" position="float"><label>Figure 1.</label><caption><title>Schematic of MPRA design.</title><p>At the top, two native genes in the genome with multiple promoter variants (stars) are shown. Below the two genes, the TSS and Upstream MPRA library designs are illustrated. The TSS library tests all variants within 144 bp of the transcription start site (dashed vertical lines), while the Upstream library tests a subset of TSS variants along with variants located further away from the transcription start site.</p></caption><graphic xlink:href="elife-62669-fig1-v2.tif" mimetype="image" mime-subtype="tiff"/></fig><fig id="fig1s1" position="float" specific-use="child-fig"><label>Figure 1—figure supplement 1.</label><caption><title>Size distribution of indels in the MPRA design.</title></caption><graphic xlink:href="elife-62669-fig1-figsupp1-v2.tif" mimetype="image" mime-subtype="tiff"/></fig><fig id="fig1s2" position="float" specific-use="child-fig"><label>Figure 1—figure supplement 2.</label><caption><title>Distribution of the number of promoter variants per gene in the MPRA design.</title></caption><graphic xlink:href="elife-62669-fig1-figsupp2-v2.tif" mimetype="image" mime-subtype="tiff"/></fig><fig id="fig1s3" position="float" specific-use="child-fig"><label>Figure 1—figure supplement 3.</label><caption><title>Schematic of the library cloning procedure.</title></caption><graphic xlink:href="elife-62669-fig1-figsupp3-v2.tif" mimetype="image" mime-subtype="tiff"/></fig><fig id="fig1s4" position="float" specific-use="child-fig"><label>Figure 1—figure supplement 4.</label><caption><title>Distributions of barcodes.</title><p>(<bold>A</bold>) Number of barcodes tagging a given oligo in the TSS library. The inset shows the range from zero to 5000 barcodes, which contains the majority of the distribution. (<bold>B</bold>) as in (<bold>A</bold>), but for the Upstream library. (<bold>C</bold>) Distribution of the number of times a given barcode was observed in the TSS annotation sequencing run. (<bold>D</bold>). As in (<bold>C</bold>), but for the Upstream library.</p></caption><graphic xlink:href="elife-62669-fig1-figsupp4-v2.tif" mimetype="image" mime-subtype="tiff"/></fig><fig id="fig1s5" position="float" specific-use="child-fig"><label>Figure 1—figure supplement 5.</label><caption><title>Number of times a given designed oligo was observed in the TSS annotation sequencing run as a function of the first two nucleotides of the oligo.</title><p>Boxplots show the median as thick horizontal line, with the box showing the 25<sup>th</sup> and 75<sup>th</sup> percentiles. Whiskers show the largest value no further than 1.5 times the inter-quartile range; data points beyond this range are shown as individual dots. Note the reduced counts for oligos starting with a ‘G’, in particular those that started with a ‘GG’.</p></caption><graphic xlink:href="elife-62669-fig1-figsupp5-v2.tif" mimetype="image" mime-subtype="tiff"/></fig><fig id="fig1s6" position="float" specific-use="child-fig"><label>Figure 1—figure supplement 6.</label><caption><title>Barcode amplification.</title><p>The figure shows details of the molecular reactions used to make Illumina sequencing libraries for barcode counting. Primer names are given in quotes. For RNA, Protocol one is on the left and Protocol two is on the right (see Materials and methods for details). Note that in Protocol 1, the PCR step can exponentially amplify only cDNA but not plasmid molecules that may have escaped DNA degradation during RNA extraction because both PCR primers bind to overhangs added in the previous steps. If, during the single extension step, ‘common_ORF_v4’ uses plasmid DNA as a template, the product lacks the p5 overhang, which contains the binding site for ‘Illumina_PCR_R’. Conversely, if during single extension, ‘RT_PCR_R_long’ (which is still present in the reaction) primes off a plasmid molecule, the product lacks the p7 overhang required by ‘RT_PCR_D7xx_F’. In protocol 2, the primers ‘common_ORF_v2’ and ‘RT_PCRPD7xx_F’ are replaced by primers ‘RT_PCR_F_long_D7xx’. These primers permit direct amplification from plasmids and from first strand cDNA but require multiple long primers for multiplexing and provide less protection against inadvertent plasmid amplification.</p></caption><graphic xlink:href="elife-62669-fig1-figsupp6-v2.tif" mimetype="image" mime-subtype="tiff"/></fig><fig id="fig1s7" position="float" specific-use="child-fig"><label>Figure 1—figure supplement 7.</label><caption><title>Reproducibility of oligo expression.</title><p>(<bold>A</bold>) Correlations of expression driven by oligos among replicates in the TSS library. (<bold>B</bold>) Average expression across TSS replicates driven by the 200 oligos from <xref ref-type="bibr" rid="bib94">Sharon et al., 2012</xref> compared to their published values. Blue points show oligos that did not include a Gal4 binding site; gray points show oligos that did include such a site. The indicated correlation was computed on oligos without Gal4 sites, because promoters with Gal4 sites are not expected to drive expression in the absence of galactose, as in the medium used here. Note that the correlation between observed and published data exists despite the different growth media between studies, and although Sharon et al. quantified expression by FACS-seq instead of our RNA-seq-based measurements. (<bold>C</bold>) Average expression driven by TSS oligos from a given gene promoter compared to that gene’s mRNA level in the genome (<xref ref-type="bibr" rid="bib4">Albert et al., 2018</xref>). TPM: transcripts per million. (<bold>D – F</bold>) as A – C, but for the Upstream library. The oligos in B and E showed higher correlations in the Upstream library than the TSS library, perhaps because the Upstream library but not the TSS library contained the same minimal <italic>HIS3</italic> promoter fragment used in Sharon et al. The correlation with native genes (<bold>C and F</bold>) was stronger in the TSS than the Upstream library, perhaps because TSS oligos more closely resembled a native promoter due to their closer proximity to the transcription start sites and the TSS library’s lack of the minimal <italic>HIS3</italic> promoter.</p></caption><graphic xlink:href="elife-62669-fig1-figsupp7-v2.tif" mimetype="image" mime-subtype="tiff"/></fig><fig id="fig1s8" position="float" specific-use="child-fig"><label>Figure 1—figure supplement 8.</label><caption><title>Correlations of expression driven by 200 oligos common to the TSS and Upstream library.</title></caption><graphic xlink:href="elife-62669-fig1-figsupp8-v2.tif" mimetype="image" mime-subtype="tiff"/></fig></fig-group><p>Massively parallel reporter assays (MPRAs) have begun to make it possible to dissect regulatory activity of DNA sequences at scale (<xref ref-type="bibr" rid="bib50">Kinney et al., 2010</xref>; <xref ref-type="bibr" rid="bib73">Melnikov et al., 2012</xref>; <xref ref-type="bibr" rid="bib77">Mulvey et al., 2020</xref>; <xref ref-type="bibr" rid="bib83">Patwardhan et al., 2009</xref>). In these approaches, which have been applied in bacteria (<xref ref-type="bibr" rid="bib16">Cambray et al., 2018</xref>; <xref ref-type="bibr" rid="bib50">Kinney et al., 2010</xref>; <xref ref-type="bibr" rid="bib55">Kosuri et al., 2013</xref>), yeast (<xref ref-type="bibr" rid="bib23">Cuperus et al., 2017</xref>; <xref ref-type="bibr" rid="bib75">Mogno et al., 2013</xref>; <xref ref-type="bibr" rid="bib94">Sharon et al., 2012</xref>), flies (<xref ref-type="bibr" rid="bib34">Gisselbrecht et al., 2013</xref>), zebrafish (<xref ref-type="bibr" rid="bib86">Rabani et al., 2017</xref>), mouse tissues (<xref ref-type="bibr" rid="bib61">Kwasnieski et al., 2014</xref>; <xref ref-type="bibr" rid="bib100">Smith et al., 2013</xref>), and cultured human cells (<xref ref-type="bibr" rid="bib77">Mulvey et al., 2020</xref>), pooled libraries of DNA oligos are placed next to or in a reporter gene and inserted into populations of cells. The activity of each oligo is then assayed in bulk by pooled high-throughput sequencing. MPRAs have been performed with synthesized libraries of designed oligos (<xref ref-type="bibr" rid="bib94">Sharon et al., 2012</xref>), with fragmented genomic DNA (<xref ref-type="bibr" rid="bib109">van Arensbergen et al., 2019</xref>; <xref ref-type="bibr" rid="bib7">Arnold et al., 2013</xref>; <xref ref-type="bibr" rid="bib111">Wang et al., 2018</xref>), and with randomly generated DNA (<xref ref-type="bibr" rid="bib23">Cuperus et al., 2017</xref>; <xref ref-type="bibr" rid="bib25">de Boer et al., 2020</xref>; <xref ref-type="bibr" rid="bib91">Rosenberg et al., 2015</xref>). MPRA readouts have included sequencing of either the oligos themselves (<xref ref-type="bibr" rid="bib7">Arnold et al., 2013</xref>) or short barcodes that tag each oligo (<xref ref-type="bibr" rid="bib60">Kwasnieski et al., 2012</xref>). MPRAs have quantified mRNA expression by sequencing cDNA (<xref ref-type="bibr" rid="bib60">Kwasnieski et al., 2012</xref>) and protein expression based on ‘FACS-Seq’ approaches in which cells are sorted into bins of increasing fluorescent reporter gene activity (<xref ref-type="bibr" rid="bib50">Kinney et al., 2010</xref>; <xref ref-type="bibr" rid="bib70">Matreyek et al., 2018</xref>; <xref ref-type="bibr" rid="bib94">Sharon et al., 2012</xref>). MPRAs have been conducted using plasmid-borne reporters as well as reporters integrated into the genome (<xref ref-type="bibr" rid="bib44">Inoue et al., 2017</xref>; <xref ref-type="bibr" rid="bib69">Maricque et al., 2019</xref>; <xref ref-type="bibr" rid="bib75">Mogno et al., 2013</xref>).</p><p>MPRAs have been used to probe DNA sequences for their ability to drive transcription (<xref ref-type="bibr" rid="bib7">Arnold et al., 2013</xref>; <xref ref-type="bibr" rid="bib48">Kheradpour et al., 2013</xref>; <xref ref-type="bibr" rid="bib111">Wang et al., 2018</xref>), dissect the importance of individual bases in regulatory elements (<xref ref-type="bibr" rid="bib83">Patwardhan et al., 2009</xref>), and examine the combined effects of multiple elements in regulatory ‘grammars’ (<xref ref-type="bibr" rid="bib24">Davis et al., 2020</xref>; <xref ref-type="bibr" rid="bib55">Kosuri et al., 2013</xref>; <xref ref-type="bibr" rid="bib75">Mogno et al., 2013</xref>; <xref ref-type="bibr" rid="bib94">Sharon et al., 2012</xref>; <xref ref-type="bibr" rid="bib100">Smith et al., 2013</xref>) in promoters (<xref ref-type="bibr" rid="bib56">Kotopka and Smolke, 2020</xref>; <xref ref-type="bibr" rid="bib66">Lubliner et al., 2015</xref>; <xref ref-type="bibr" rid="bib94">Sharon et al., 2012</xref>; <xref ref-type="bibr" rid="bib114">Weingarten-Gabbay et al., 2019</xref>), UTRs (<xref ref-type="bibr" rid="bib23">Cuperus et al., 2017</xref>; <xref ref-type="bibr" rid="bib27">Dvir et al., 2013</xref>; <xref ref-type="bibr" rid="bib86">Rabani et al., 2017</xref>; <xref ref-type="bibr" rid="bib93">Shalem et al., 2015</xref>) and enhancers (<xref ref-type="bibr" rid="bib7">Arnold et al., 2013</xref>; <xref ref-type="bibr" rid="bib54">Klein et al., 2019</xref>; <xref ref-type="bibr" rid="bib73">Melnikov et al., 2012</xref>; <xref ref-type="bibr" rid="bib83">Patwardhan et al., 2009</xref>). Other applications have assayed sequences that promote splicing (<xref ref-type="bibr" rid="bib19">Cheung et al., 2019</xref>; <xref ref-type="bibr" rid="bib91">Rosenberg et al., 2015</xref>), translation (<xref ref-type="bibr" rid="bib35">Goodman et al., 2013</xref>; <xref ref-type="bibr" rid="bib113">Weingarten-Gabbay et al., 2016</xref>), DNA methylation (<xref ref-type="bibr" rid="bib57">Krebs et al., 2014</xref>) and RNA editing (<xref ref-type="bibr" rid="bib92">Safra et al., 2017</xref>). More recently, MPRAs have identified individual human DNA variants that alter gene expression (<xref ref-type="bibr" rid="bib107">Tewhey et al., 2016</xref>; <xref ref-type="bibr" rid="bib108">Ulirsch et al., 2016</xref>) in studies ranging in scale from variants in specific regions implicated by genome-wide association studies for a given disease (<xref ref-type="bibr" rid="bib20">Choi et al., 2019</xref>; <xref ref-type="bibr" rid="bib64">Liu et al., 2017</xref>; <xref ref-type="bibr" rid="bib82">Pashos et al., 2017</xref>; <xref ref-type="bibr" rid="bib110">Vockley et al., 2015</xref>) to a genome-wide survey of nearly six million common single-nucleotide polymorphisms (<xref ref-type="bibr" rid="bib109">van Arensbergen et al., 2019</xref>). In spite of these successes, the size of the human genome, which harbors tens of millions of rare as well as common variants (<xref ref-type="bibr" rid="bib1">1000 Genomes Project Consortium et al., 2015</xref>), combined with a high degree of tissue-specificity in gene expression and the activity of regulatory DNA <xref ref-type="bibr" rid="bib36">GTEx Consortium et al., 2017</xref>; <xref ref-type="bibr" rid="bib45">Inoue et al., 2019</xref>; <xref ref-type="bibr" rid="bib69">Maricque et al., 2019</xref> have complicated dissection of causal variants in local eQTLs. The compact gene-regulatory regions of <italic>S. cerevisiae</italic>, combined with comprehensive eQTL maps (<xref ref-type="bibr" rid="bib4">Albert et al., 2018</xref>), provide an excellent opportunity to study regulatory variants systematically.</p><p>Here, we used an MPRA to probe thousands of intergenic variants that differ between two yeast isolates. We identified 451 variants with significant <italic>cis</italic>-acting effects on mRNA expression. These causal variants underlie known local eQTLs. We found that individual local eQTLs can harbor multiple causal variants, including pairs of variants with non-additive, epistatic effects. Causal variants tended to alter transcription factor binding motifs and showed signs of evolving under negative selection. Combinations of these features predicted causal variants better than expected by chance, albeit with modest accuracy.</p></sec><sec sec-type="results" id="s2"><title>Results</title><sec id="s2-1"><title>An MPRA to assay <italic>cis</italic>-regulatory variants in yeast promoters</title><p>At least half of the genes in the yeast genome are influenced by local eQTLs that segregate between the laboratory strain BY, a close relative of the genome reference strain S288C, and RM, a vineyard isolate that is related to strains commonly used in wine making (<xref ref-type="bibr" rid="bib85">Peter et al., 2018</xref>). Most of these eQTLs act in <italic>cis</italic> (<xref ref-type="bibr" rid="bib4">Albert et al., 2018</xref>), making the BY and RM strains a rich reservoir for identifying causal <italic>cis</italic>-acting variants. Here, we studied DNA variants in yeast promoters, which we defined as the intergenic region upstream of the transcription start site of a given gene up to the coding region of the adjacent gene, for a maximum of 1000 bases. These promoter regions differ between BY and RM at 11,768 single-nucleotide variants (SNVs) and 2442 insertion/deletion variants (indels) in 3176 genes (<xref ref-type="bibr" rid="bib10">Bloom et al., 2013</xref>).</p><p>To assay the effects of individual variants on gene expression, we designed an MPRA composed of two synthetic promoter libraries encoded by pooled oligonucleotides (<xref ref-type="fig" rid="fig1">Figure 1</xref>, <xref ref-type="table" rid="table1">Table 1</xref>, <xref ref-type="supplementary-material" rid="supp1">Supplementary file 1</xref>). The ‘TSS’ library assayed all variants in the 144 nucleotides immediately upstream of the transcription start site (<xref ref-type="bibr" rid="bib84">Pelechano et al., 2013</xref>), a region that is highly enriched for transcription factor binding sites (<xref ref-type="bibr" rid="bib63">Lin et al., 2010</xref>). When multiple variants were present within this region for a given gene, we designed one sequence that carried the BY allele at all variants, one sequence that carried the RM allele at all variants, and a set of sequences that each carried the RM allele at a single variant and the BY allele at all other variants. The ‘Upstream’ library mostly assayed variants located further upstream of the transcription start site than those in the TSS library, using a pair of sequences that represented 144 bp of genomic DNA centered on the given variant. By design, the TSS and Upstream libraries shared a subset of variants. Together, they assayed 7005 unique variants (5,758 SNVs and 1247 indels; <xref ref-type="fig" rid="fig1s1">Figure 1—figure supplement 1</xref>) in the promoters of 3076 genes, with a median of two variants per gene (<xref ref-type="fig" rid="fig1s2">Figure 1—figure supplement 2</xref>).</p><table-wrap id="table1" position="float"><label>Table 1.</label><caption><title>Library design and representation in experiments.</title><p><supplementary-material id="table1sdata1"><label>Table 1—source data 1.</label><caption><title>Information on replicate samples.</title></caption><media xlink:href="elife-62669-table1-data1-v2.xlsx" mimetype="application" mime-subtype="xlsx"/></supplementary-material></p></caption><table frame="hsides" rules="groups"><thead><tr><th valign="top">Library</th><th valign="top">TSS</th><th valign="top">Upstream</th></tr></thead><tbody><tr><td valign="top">Designed oligos</td><td valign="top">7211</td><td valign="top">9882</td></tr><tr><td valign="top">Variants in design</td><td valign="top">3645</td><td valign="top">4547</td></tr><tr><td valign="top">Genes in design</td><td valign="top">2172</td><td valign="top">1918</td></tr><tr><td valign="top">Oligos in finished library</td><td valign="top">6565</td><td valign="top">9646</td></tr><tr><td valign="top">Barcodes</td><td valign="top">9.2 million</td><td valign="top">20 million</td></tr><tr><td valign="top">Median barcodes per oligo</td><td valign="top">590</td><td valign="top">1008</td></tr><tr><td valign="top">Variants with data</td><td valign="top">2427</td><td valign="top">4467</td></tr><tr><td valign="top">Genes with data</td><td valign="top">1429</td><td valign="top">1824</td></tr><tr><td valign="top">Number of replicates</td><td valign="top">12</td><td valign="top">6</td></tr></tbody></table></table-wrap><p>The synthesized oligo libraries were placed upstream of a yEGFP reporter gene (<xref ref-type="bibr" rid="bib95">Sheff and Thorn, 2004</xref>) on a low-copy number yeast plasmid (<xref ref-type="fig" rid="fig1">Figure 1</xref>). Prior to adding the yEGFP gene, we added random barcodes with a length of 20 nucleotides downstream of the reporter gene (<xref ref-type="fig" rid="fig1">Figure 1</xref> and <xref ref-type="fig" rid="fig1s3">Figure 1—figure supplement 3</xref>). These barcodes were expressed as part of the 3’ end of the reporter mRNA, such that barcode abundance provided a measure of gene expression driven by the given oligo. We used paired-end sequencing to map each barcode to the oligo it tagged (<xref ref-type="fig" rid="fig1s3">Figure 1—figure supplement 3</xref>). Most oligos were tagged by hundreds of barcodes (<xref ref-type="table" rid="table1">Table 1</xref>, <xref ref-type="fig" rid="fig1s4">Figure 1—figure supplement 4</xref>) to control for possible influences of the expressed barcodes on mRNA levels. The final plasmid libraries we used in our experiments contained more than 90% of the designed oligos, which assayed 5832 variants in the promoters of 2503 genes (<xref ref-type="table" rid="table1">Table 1</xref>; see Materials and methods and <xref ref-type="fig" rid="fig1s5">Figure 1—figure supplement 5</xref> for a description of oligos lost during oligo synthesis and/or cloning). The libraries were transformed into the BY strain and grown in independent replicate cultures (<xref ref-type="table" rid="table1">Table 1</xref>). We grew each culture to late exponential phase, extracted mRNA and plasmid DNA from each culture, and sequenced barcodes (<xref ref-type="fig" rid="fig1s6">Figure 1—figure supplement 6</xref>) to a median depth of 46 million RNA reads (range 19–70 million) and 31 million DNA reads (range 14–84 million) per sample (<xref ref-type="supplementary-material" rid="table1sdata1">Table 1—source data 1</xref>).</p></sec><sec id="s2-2"><title>Oligo expression is reproducible and reflects gene expression in the genome</title><p>We conducted three analyses to assess the reliability of our data. First, we measured the reproducibility of oligo-driven reporter expression. We summed the RNA counts of all barcodes assigned to a given oligo (<xref ref-type="supplementary-material" rid="supp2">Supplementary file 2</xref>), divided them by the respective summed DNA counts to normalize for unequal library composition, and log<sub>2</sub>-transformed the resulting ratio. Spearman’s rank correlation coefficients (rho) among pairs of replicates ranged from 0.79 to 0.90 (median = 0.83; all p&lt;2.2e-16) in the Upstream library and from 0.55 to 0.93 (median = 0.74; all p&lt;2.2e-16) in the TSS library (<xref ref-type="fig" rid="fig1s7">Figure 1—figure supplement 7A &amp; D</xref>).</p><p>Second, our two libraries included a common set of 200 oligos whose sequences we had sampled from a prior study (<xref ref-type="bibr" rid="bib94">Sharon et al., 2012</xref>), allowing us to determine reproducibility between replicates, libraries, and with prior work. Median correlations between TSS and Upstream replicates (rho = 0.76; all p&lt;2.2e-16) were nearly as high as those between replicates within each library (TSS: 0.8, Upstream: 0.99; all p&lt;2.2e-16; <xref ref-type="fig" rid="fig1s8">Figure 1—figure supplement 8</xref>). Further, the 200 oligos showed significant, positive correlations with their published expression levels (rho ≥ 0.4, p≤1e-6), in spite of experimental and design differences between studies (<xref ref-type="fig" rid="fig1s7">Figure 1—figure supplement 7B &amp; E</xref>).</p><p>Third, we asked if the promoter fragments in our plasmid libraries were able to recapitulate the expression of genes in their native genomic locations. To do so, we computed the average expression driven by all oligos extracted from the promoter region of a given gene. Although each oligo contained at most 150 bp of promoter sequence on a plasmid, expression in both libraries correlated significantly (rho ≥ 0.23, p&lt;2.2e-16) with the native mRNA levels of genes in the genome (<xref ref-type="fig" rid="fig1s7">Figure 1—figure supplement 7C &amp; F</xref>). In sum, our assay quantified the regulatory activity of promoter fragments in a manner that was reproducible within our study and when compared to earlier work, and that reflected the activity of native promoters in the genome.</p></sec><sec id="s2-3"><title>Identification of hundreds of <italic>cis</italic>-acting promoter variants</title><p>The median fold change between alleles was 1.4-fold, and only 29 causal variants (6%) altered expression by more than two-fold (<xref ref-type="fig" rid="fig2">Figure 2A</xref>). These magnitudes are in good agreement with known genetic effects in the BY/RM strains, in which the vast majority of local eQTLs (<xref ref-type="bibr" rid="bib4">Albert et al., 2018</xref>) and allele-specific, <italic>cis</italic>-acting effects on expression (<xref ref-type="bibr" rid="bib2">Albert et al., 2014a</xref>) alter mRNA abundance by less than 2-fold.</p><fig-group><fig id="fig2" position="float"><label>Figure 2.</label><caption><title>Identification of single causal variants.</title><p>(<bold>A</bold>) A scatterplot showing the effect size and significance for each variant. The genome-wide significance threshold is shown as a dashed horizontal line. Variants with the most significant effects, along with the genes they affect, are indicated. Variants and genes highlighted in color are described in the text. The histograms at the top shows the distribution of effect sizes for causal (blue) and non-causal (gray) variants. (<bold>B</bold>) A variant in the promoter of OLE1 known to affect OLE1 expression has a significant effect in the Upstream MPRA. The figure shows expression values for oligos carrying the two alleles. Colored lines and dots indicate different biological replicate experiments. Boxplots show the median as thick line, with the box showing the 25th and 75th percentiles. Whiskers show the largest value no further than 1.5 times the inter-quartile range; points beyond this range are shown as individual points. (<bold>C</bold>) As in (<bold>B</bold>) for a variant in the SFA1 promoter, which was detected in the TSS library. To identify individual causal variants, we tested each promoter variant for its effect on reporter gene expression. We detected 166 variants with significant effects in the TSS library and 293 variants in the Upstream library at a false discovery rate (FDR) of 5% (<xref ref-type="supplementary-material" rid="fig2sdata1">Figure 2—source data 1</xref>). The π<sub>1</sub> statistic (<xref ref-type="bibr" rid="bib104">Storey and Tibshirani, 2003</xref>) computed across all variants suggested that at least 26 and 31% of variants had effects on gene expression in the TSS and Upstream libraries, respectively, even if these variants could not all be detected with individual significance. There were 451 unique variants that reached significance across the two partially overlapping libraries (A, <xref ref-type="supplementary-material" rid="fig2sdata2">Figure 2—source data 2</xref>).</p><p><supplementary-material id="fig2sdata1"><label>Figure 2—source data 1.</label><caption><title>Statistical tests for each variant.</title><p>Results for the TSS and Upstream libraries are in separate worksheets.</p></caption><media xlink:href="elife-62669-fig2-data1-v2.xlsx" mimetype="application" mime-subtype="xlsx"/></supplementary-material></p><p><supplementary-material id="fig2sdata2"><label>Figure 2—source data 2.</label><caption><title>Statistical results for each variant after aggregating across the two sub libraries.</title></caption><media xlink:href="elife-62669-fig2-data2-v2.xlsx" mimetype="application" mime-subtype="xlsx"/></supplementary-material></p></caption><graphic xlink:href="elife-62669-fig2-v2.tif" mimetype="image" mime-subtype="tiff"/></fig><fig id="fig2s1" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 1.</label><caption><title>Reproducibility of variant effects.</title><p>(<bold>A</bold>) Variants measured in the same library (TSS or Upstream) but in opposite orientation (‘strand’) with respect to the reporter gene. Blue points: variants with significant (5% FDR) effects on the plus strand. Red points: variants with significant effects on the minus strand. Purple, larger points: variants significant in both orientations. The indicated correlations were computed for variants that were significant in at least one of the two orientations. (<bold>B</bold>) As in (<bold>A</bold>) but comparing effects of variants measured on the same strand but in the two different libraries.</p></caption><graphic xlink:href="elife-62669-fig2-figsupp1-v2.tif" mimetype="image" mime-subtype="tiff"/></fig></fig-group><p>Our MPRA assayed some variants multiple times in different sequence contexts, and we used this redundancy to assess the reproducibility of variant effects. First, 359 variants that reside in the intergenic region between two divergently expressed genes were tested twice in the same library, but in opposite orientation relative to the reporter gene (<xref ref-type="fig" rid="fig2s1">Figure 2—figure supplement 1A</xref>). Variants that were significant in at least one strand orientation were likely to also be significant in the other orientation (Fisher’s exact test (FET): p=0.0003, odds ratio (OR) = 7). These significant variants also agreed well in the direction of variant effect, that is whether the RM allele drives higher or lower expression than the BY allele (FET: p=0.007, OR = 6.1). Second, 527 variants were assayed in both the TSS and the Upstream libraries, where they were embedded in oligos that had the same strand orientation but differed in the exact window of DNA surrounding each variant (<xref ref-type="fig" rid="fig2s1">Figure 2—figure supplement 1B</xref>). Among these, variants that were significant in at least one library were likely to also be significant in the other (FET p=0.007, OR = 4), with significant directional agreement (FET p=0.046, OR = 2.9). Thus, in spite of the small effects of natural sequence variants, our assay was able to reproducibly identify individual causal variants.</p><p>Our assay identified several known causal variants. For example, a variant in the promoter of the <italic>OLE1</italic> gene affects <italic>OLE1</italic> expression in <italic>cis</italic> and is likely to be the single causal variant in this promoter (<xref ref-type="bibr" rid="bib67">Lutz et al., 2019</xref>). This variant was highly significant in the MPRA (<xref ref-type="fig" rid="fig2">Figure 2B</xref>), while a second variant in the <italic>OLE1</italic> promoter was not (<xref ref-type="fig" rid="fig2">Figure 2A</xref>).</p><p>In the promoter of the <italic>SFA1</italic> gene, the MPRA detected a single causal variant out of nine that were assayed (<xref ref-type="fig" rid="fig2">Figure 2A</xref>). The RM allele of the causal variant is also present in a strain used for bioethanol production (YJS329), where it has been proposed to increase <italic>SFA1</italic> expression by creating an Msn2/4 binding site (<xref ref-type="bibr" rid="bib71">Maurer et al., 2017</xref>; <xref ref-type="bibr" rid="bib121">Zheng et al., 2012</xref>). In our assay, this allele increased gene expression significantly (<xref ref-type="fig" rid="fig2">Figure 2C</xref>). Taken together, these results show that our MPRA reproducibly detected hundreds of individual <italic>cis</italic>-acting DNA variants.</p></sec><sec id="s2-4"><title>Local eQTLs can be caused by single or multiple causal variants</title><p>The identity of causal variants in QTL regions is a major question in the genetics of complex traits. We asked if the variants identified by our MPRA underlie the many local eQTLs that segregate between the BY and RM strains. Specifically, we turned to local eQTLs mapped in 1012 recombinant individuals obtained by crossing BY and RM (<xref ref-type="bibr" rid="bib4">Albert et al., 2018</xref>). Because of genetic linkage in the cross, the effects of neighboring variants create one aggregated, spread-out signal of local variation at each gene. For 2884 genes, these effects were strong enough to be detected as local eQTLs at genome-wide significance (<xref ref-type="bibr" rid="bib4">Albert et al., 2018</xref>). We asked to what extent these local eQTLs can be explained by individual causal variants identified here (<xref ref-type="supplementary-material" rid="fig3sdata1">Figure 3—source data 1</xref>).</p><p>Initially, we considered all 2300 genes with available data from both the cross and the MPRA, irrespective of whether their local effect had reached significance in the cross. Likewise, we summed the MPRA effects of all assayed variants for a given gene, irrespective of their significance. There was a significant correlation between local eQTL effects and summed MPRA effects (rho = 0.1, p=1e-6). This overall correlation is likely degraded by noise in both datasets, in particular for small, non-significant effects. Therefore, we first restricted our analyses to variants with significant (5% FDR) MPRA effects. The summed effects of these significant variants showed a stronger correlation with eQTL effects than did those of all variants (rho = 0.24, p=5e-6). Next, to also avoid noise in the eQTL effect estimates, we further restricted the comparison to the 238 genes with strong local eQTLs—those with a LOD score of at least 50. This filter again improved the correlation with summed significant MPRA effects (rho = 0.47, p=2e-5). Allele-specific expression in diploid hybrids provides an independent estimate of the aggregated effect of all <italic>cis</italic>-acting variants that affect a given gene (<xref ref-type="bibr" rid="bib28">Emerson et al., 2010</xref>; <xref ref-type="bibr" rid="bib89">Ronald et al., 2005</xref>; <xref ref-type="bibr" rid="bib117">Wittkopp et al., 2004</xref>). Genes with significant allele-specific expression in published BY/RM hybrid data (<xref ref-type="bibr" rid="bib2">Albert et al., 2014a</xref>; <xref ref-type="bibr" rid="bib4">Albert et al., 2018</xref>) were more likely to have a causal variant in our MPRA (FET: OR = 3.2, p&lt;2.2e-16). These agreements show that, in aggregate, our MPRA successfully captured effects of variants that cause local, <italic>cis</italic>-acting eQTLs.</p><p>We asked whether single MPRA variants could account for local eQTLs. At each gene, we extracted the most significant variant among those with an FDR of ≤5%. The effects of these single variants correlated with strong local eQTLs (rho = 0.41, p=3e-4) and showed significant agreement between the direction of the effects of the variants and the eQTLs (FET: OR = 6.9, p=9e-5; <xref ref-type="fig" rid="fig3">Figure 3A</xref>). For example, the MPRA effect for the single causal variant in the <italic>OLE1</italic> promoter (<xref ref-type="bibr" rid="bib67">Lutz et al., 2019</xref>) recapitulated the magnitude of the <italic>OLE1</italic> local eQTL almost perfectly (<xref ref-type="fig" rid="fig3">Figure 3A &amp; C</xref>). Thus, we identified individual causal variants that underlie local eQTLs.</p><fig id="fig3" position="float"><label>Figure 3.</label><caption><title>Comparison of variant effects to local eQTLs.</title><p>(<bold>A</bold>) Scatterplot showing the MPRA effect of the most significant causal variant per gene (y-axis) versus the effect of the local eQTL (x-axis). Red dots indicate local eQTLs with a LOD score of at least 50. Genes describes in the text and in panel (<bold>C</bold>) are highlighted in bold. Other genes also present in panel (<bold>B</bold>) are indicated in regular font. The dashed diagonal shows the case of equal MPRA and local eQTL effects. Dashed horizontal and vertical lines indicate no effect. The panel only shows variants with a significant MPRA effect. (<bold>B</bold>) As in (<bold>A</bold>), but for the second-most significant causal variant per gene. Note the different scale of the y-axis between A and B. (<bold>C</bold>). Examples of summed MPRA effects of individual variants compared to the local eQTL for the given gene. (<bold>D</bold>). Spearman correlation coefficients MPRA variants versus local eQTLs as a function of whether a variant is bound by a nucleosome in the genome. Significance of the correlation is indicated. Error bars show 95% confidence intervals for the strength of the correlation.</p><p><supplementary-material id="fig3sdata1"><label>Figure 3—source data 1.</label><caption><title>Local eQTLs and variant results.</title></caption><media xlink:href="elife-62669-fig3-data1-v2.xlsx" mimetype="application" mime-subtype="xlsx"/></supplementary-material></p></caption><graphic xlink:href="elife-62669-fig3-v2.tif" mimetype="image" mime-subtype="tiff"/></fig><p>QTL regions for organismal traits can harbor multiple causal variants (<xref ref-type="bibr" rid="bib68">Mackay et al., 2009</xref>), but it is less clear whether QTLs for cellular traits such as gene expression have a similarly complex molecular basis. To test whether local eQTLs contain multiple causal variants, we considered, where available, the second-most significant MPRA variant for each gene, if that variant was still significant at an FDR of 5%. The effects of these second variants correlated with eQTL effects (rho = 0.3, p=0.03; <xref ref-type="fig" rid="fig3">Figure 3B</xref>). Further, there were significant correlations between the number of nominally significant (p&lt;0.05) MPRA variants identified for a gene and that gene’s local eQTL LOD score (rho = 0.12, p=0.003) and its absolute eQTL effect size (rho = 0.14, p=3e-5). Thus, some local eQTLs are due to multiple causal promoter variants.</p><p>Some genes were affected by more than two causal variants. For example, we detected four causal variants at the <italic>CWP1</italic> gene (<xref ref-type="fig" rid="fig2">Figure 2A</xref>). At each of these variants, the RM allele increased expression, and their summed effects approximated that of the <italic>CWP1</italic> local eQTL (<xref ref-type="fig" rid="fig3">Figure 3C</xref>).</p><p>Some genes had two significant variants with opposite direction of effect. For example, we detected two variants located 47 bases apart in the promoter of the <italic>UFO1</italic> gene that had effects of similar magnitude but in opposite direction (<xref ref-type="fig" rid="fig2">Figure 2A</xref>). <italic>UFO1</italic> does not have a significant local eQTL (<xref ref-type="fig" rid="fig3">Figure 3C</xref>), suggesting that this gene is influenced by two variants whose effects cancel each other out.</p><p>Together, these observations suggest that the molecular basis of local eQTLs spans a range of scenarios. Some local eQTLs are due to single causal variants, while others arise from joint effects of multiple causal variants. When multiple variants have effects in the same direction, they sum to create a stronger eQTL. When their effects are in the opposite directions, they may be invisible in a cross.</p></sec><sec id="s2-5"><title>The effects of individual variants may be influenced by nucleosome binding</title><p>Our plasmid-based MPRA tested variants in a molecular context that differs from the native genomic context in two respects: in the DNA sequence beyond that present on the oligos and, potentially, in the chromatin state in which each variant is embedded. Specifically, nucleosomes influence gene expression (<xref ref-type="bibr" rid="bib38">Han and Grunstein, 1988</xref>), and may mask or otherwise alter the effects of variants they bind (<xref ref-type="bibr" rid="bib87">Rando and Winston, 2012</xref>; <xref ref-type="bibr" rid="bib124">Zhu et al., 2018</xref>). We asked whether the effects of variants located in nucleosome-free regions in the genome (<xref ref-type="bibr" rid="bib15">Brogaard et al., 2012</xref>) were better captured by our MPRA than those of nucleosome-bound variants. Indeed, the summed effects of significant variants in nucleosome-free regions correlated better with strong local eQTLs than those of nucleosome-bound variants, which showed no significant correlation (<xref ref-type="fig" rid="fig3">Figure 3D</xref>). A similar difference was seen for the single most significant MPRA variant (<xref ref-type="fig" rid="fig3">Figure 3D</xref>). This better agreement between the genome and the MPRA for the effects of variants in nucleosome-free regions would be expected if these variants are also not bound by nucleosomes on the plasmids, and are thus exposed to a molecular environment that resembles that in the genome. More broadly, this result suggests that nucleosome occupancy and chromatin state may alter the regulatory effects of genomic variants, consistent with MPRAs in human cells (<xref ref-type="bibr" rid="bib45">Inoue et al., 2019</xref>; <xref ref-type="bibr" rid="bib69">Maricque et al., 2019</xref>) and with the fact that many human local eQTLs vary among tissues and cell types (<xref ref-type="bibr" rid="bib36">GTEx Consortium et al., 2017</xref>; <xref ref-type="bibr" rid="bib119">Yao et al., 2020</xref>).</p></sec><sec id="s2-6"><title>Non-additive interactions among promoter variants</title><p>The presence of multiple causal variants in single promoters raises the question of whether these variants have independent, additive effects or if the effects of some variants are modulated by the presence of the other variant(s) in a non-additive, epistatic interaction. To address this question, we made use of the fact that the TSS library included 342 variant pairs for which there were oligos representing all four combinations of the two alleles at the two variants. For each such variant pair, we tested whether the joint effect of both variants differed from the sum of their individual effects (<xref ref-type="supplementary-material" rid="fig4sdata1">Figure 4—source data 1</xref>).</p><p>There were five variant pairs that showed evidence for non-additive effects at an FDR of ≤ 20%. While the π<sub>1</sub> statistic (<xref ref-type="bibr" rid="bib104">Storey and Tibshirani, 2003</xref>) suggested that at least 34% of variant pairs have non-additive effects, the majority of these cases could not be detected at individual significance, presumably due to limited statistical power.</p><p>The five significant pairs showed a range of epistatic patterns (<xref ref-type="fig" rid="fig4">Figure 4</xref> and <xref ref-type="fig" rid="fig4s1">Figure 4—figure supplement 1</xref>). For example, the RM alleles at two variants upstream of <italic>DAD2</italic> did not have significant effects individually (p&gt;0.4 compared to the oligo with two BY alleles), but when combined, the two RM alleles drove significantly higher expression (p=2e-6, FDR = 0.01%; <xref ref-type="fig" rid="fig4">Figure 4</xref>). RM carries a derived allele at each of these variants, indicated by the fact that the BY alleles are present in a strain from the Taiwanese clade of yeast isolates, which split early from other isolates (including BY and RM) during <italic>S. cerevisiae</italic> evolution (<xref ref-type="bibr" rid="bib85">Peter et al., 2018</xref>). Both derived alleles are found at high and nearly identical frequencies in the yeast population (~74%), and the two variants, which are separated by only two nucleotides, show very high linkage disequilibrium (D’=0.999, r<sup>2</sup> = 0.99). Thus, these two epistatic variants form a tightly linked haplotype that increases gene expression only when both RM alleles are present, perhaps because both variants are needed to create a binding site for a transcriptional activator.</p><fig-group><fig id="fig4" position="float"><label>Figure 4.</label><caption><title>Epistasis among promoter variants.</title><p>Each panel shows, for one gene, MPRA expression driven by four oligos with the indicated combination of BY and RM alleles at the two variants. Each panel states the gene name in bold along with multiple-testing adjusted and raw interaction p-values. Variants are given as “chromosome:position reference/alternative allele”. Colored lines between boxplots connect the data for a given oligo in the different biological replicates. Boxplots show the median as thick line, with the box showing the 25th and 75th percentiles. Whiskers show the largest value no further than 1.5 times the inter-quartile range; any observations beyond this range are shown as individual points. (<bold>A</bold>). Promoter variants for DAD2. (<bold>B</bold>) Promoter variants for RTT101.</p><p><supplementary-material id="fig4sdata1"><label>Figure 4—source data 1.</label><caption><title>Results from the test for epistatic interactions.</title></caption><media xlink:href="elife-62669-fig4-data1-v2.xlsx" mimetype="application" mime-subtype="xlsx"/></supplementary-material></p></caption><graphic xlink:href="elife-62669-fig4-v2.tif" mimetype="image" mime-subtype="tiff"/></fig><fig id="fig4s1" position="float" specific-use="child-fig"><label>Figure 4—figure supplement 1.</label><caption><title>Additional cases of significant epistasis between promoter variants.</title><p>Each panel shows, for one gene, MPRA expression driven by four oligos with the indicated combination of BY and RM alleles at the two variants. Each panel states the gene name in bold along with multiple-testing adjusted as well as raw interaction p-values. Variants are given as ‘chromosome:position reference/alternative allele’. Colored lines between boxplots connect the data for a given oligo in the different biological replicates. Boxplots show the median as thick line, with the box showing the 25<sup>th</sup> and 75<sup>th</sup> percentiles. Whiskers show the largest value no further than 1.5 times the inter-quartile range; any observations beyond this range are shown as individual points.</p></caption><graphic xlink:href="elife-62669-fig4-figsupp1-v2.tif" mimetype="image" mime-subtype="tiff"/></fig></fig-group><p>The RM alleles at two SNVs upstream of <italic>RTT101</italic> each reduced expression compared to the BY background (FDR &lt; 0.1%), but an oligo carrying both RM alleles reduced expression to a similar degree as either variant alone (<xref ref-type="fig" rid="fig4">Figure 4B</xref>). Both RM alleles are derived, and the two variants have very little recombination (D’=0.998). However, the two RM alleles do not form a common haplotype (r<sup>2</sup> = 0.08), likely because while the RM ‘A’ allele at 352,322 bp on chromosome X is present in 42% of yeast isolates, the RM ‘G’ allele at 352,304 bp is much rarer, with a frequency of 6%. These results suggest that a common promoter variant reduces the expression of <italic>RTT101</italic>. A second, much rarer variant has a similar effect in isolation, but its effect is obscured in the presence of the more common variant, perhaps because both variants abrogate binding of the same transcriptional activator.</p><p>Together, these results show that <italic>cis</italic>-acting promoter variants can influence gene expression in a non-additive fashion. Such epistatic effects may be widespread, but the resultant deviations from additive expectations are often small, making their systematic detection challenging.</p></sec><sec id="s2-7"><title>Characteristics of causal variants</title><p>The identification of hundreds of causal variants enabled us to ask whether causal variants tend to share molecular or population genetic characteristics. To address this question, we assembled a set of 2998 features describing each variant. These features comprised sequence characteristics of the alleles, evolutionary properties of the variant, and features that describe the gene that is regulated by the promoter in which the variant resides (<xref ref-type="supplementary-material" rid="supp3">Supplementary file 3</xref> and <xref ref-type="supplementary-material" rid="supp4">Supplementary file 4</xref>). Most features (2,967) described transcription factor (TF) binding (see below).</p><p>We divided variants into a ‘causal’ group (defined as variants with an FDR of ≤5%; n = 459) and a non-causal group (variants with a raw p-value&gt;0.2; n = 3,774). For variants located in divergent promoters that were assayed in both orientations, the results from the two orientations were considered separately. For this reason, the number of causal variants in these analyses was slightly higher than the 451 unique variants reported above. We used logistic regression to test the association of each feature with variant causality (<xref ref-type="supplementary-material" rid="fig5sdata1">Figure 5—source data 1</xref>). While these analyses cannot isolate the individual contributions of features that are correlated with each other, they provide an overview of the characteristics of causal variants.</p><p>Overall, causal variants were more likely to be SNVs than indels (<xref ref-type="fig" rid="fig5">Figure 5A</xref>). At the same time, longer indels were more likely to be causal than shorter indels (<xref ref-type="fig" rid="fig5">Figure 5A</xref>). Causal variants were less likely to be bound by a nucleosome in the genome (FDR = 4%). Nucleosomes may shield such variants from DNA-binding factors even on the plasmid reporter if the local DNA sequence around the tested variants can direct nucleosome formation at least partially.</p><fig-group><fig id="fig5" position="float"><label>Figure 5.</label><caption><title>Association of features with variant causality.</title><p>(<bold>A</bold>) Non-TF features. The figure shows the strength of association between each feature and variant causality. Error bars show the standard error of the mean. Significant associations are indicated by three stars (FDR &lt; 5%) or one star (nominal p-value&lt;0.05). Non-significant features are shown in lighter coloring. (<bold>B</bold>) TF summary features aggregated across all 196 TFs, separated by strand, mode of aggregation across TFs, strength of binding (weak or strong), and mode of comparing allelic PWM scores across sliding sequence windows spanning each variant (Materials and methods). Each of these summary features was significantly associated with variant causality at an FDR of &lt;5%. (<bold>C</bold>) Distributions of logistic regression estimates for strong vs. weak binding for individual TFs. The p-value shows the result of a Wilcoxon rank test. See also <xref ref-type="fig" rid="fig5s1">Figure 5—figure supplement 1</xref>.</p><p><supplementary-material id="fig5sdata1"><label>Figure 5—source data 1.</label><caption><title>Single-feature regression analyses of causal variants.</title><p>Logistic and linear regression (to predict log-fold change) are in separate worksheets.</p></caption><media xlink:href="elife-62669-fig5-data1-v2.xlsx" mimetype="application" mime-subtype="xlsx"/></supplementary-material></p></caption><graphic xlink:href="elife-62669-fig5-v2.tif" mimetype="image" mime-subtype="tiff"/></fig><fig id="fig5s1" position="float" specific-use="child-fig"><label>Figure 5—figure supplement 1.</label><caption><title>Examples of predicted TFBS changes at individual variants.</title><p>(<bold>A</bold>) A variant in the promoter of the <italic>SFA1</italic> gene alters a strong Msn2/4 motif (Yeastract consensus motif: CCCCT); as well as a strong Haa1 motif (SMGGSG). TFBSs detected by the Yeastract website in the given sequence are shown as colored arrows indicating the strand on which the motif match was detected. TFBSs above the sequence were detected for the BY allele, while those below the sequence were detected for the RM allele. Two TFBSs that differ between the alleles are shown as stronger lines. The histogram shows the distribution of change in TFBS score to weak TFBSs. The dashed yellow line indicates the median change in TFBS score. TFBS scores correspond to log<sub>2</sub>(predicted likelihood of binding). (<bold>B</bold>) A variant in the promoter of the <italic>HAL5</italic> gene that did not alter strong TFBSs. Nevertheless, this variant caused a similar change in gene expression, accompanied by a similar median change to weak TFBSs, as the variant in (<bold>A</bold>).</p></caption><graphic xlink:href="elife-62669-fig5-figsupp1-v2.tif" mimetype="image" mime-subtype="tiff"/></fig></fig-group><p>Causal variants were enriched at nucleotides with higher PhastCons scores (FDR = 1%, <xref ref-type="fig" rid="fig5">Figure 5A</xref>), suggesting that natural variants that occur at nucleotides that are more conserved across yeast species are more likely to affect gene expression than those at less conserved nucleotides. Variants in the promoters of essential genes were less likely to be causal (FDR = 6%). Causal variants were also less likely to occur in promoters of genes with many synthetic genetic interactions (<xref ref-type="fig" rid="fig5">Figure 5A</xref>), which tends to be a property of essential genes (<xref ref-type="bibr" rid="bib22">Costanzo et al., 2016</xref>). Essential genes are required for yeast viability, and are known to have less genetic (<xref ref-type="bibr" rid="bib4">Albert et al., 2018</xref>), environmental (<xref ref-type="bibr" rid="bib29">Eng et al., 2010</xref>), and stochastic (<xref ref-type="bibr" rid="bib80">Newman et al., 2006</xref>) variation in gene expression. The observation that natural variants in promoters of essential genes are less likely to perturb gene expression is consistent with negative selection acting to purge causal variants from essential gene promoters. In further support of this hypothesis, causal variants had lower derived allele frequencies than non-causal variants (FDR = 10%), as expected if negative selection prevents variants that affect gene expression from rising to higher frequency (<xref ref-type="bibr" rid="bib53">Kita et al., 2017</xref>; <xref ref-type="bibr" rid="bib90">Ronald and Akey, 2007</xref>).</p></sec><sec id="s2-8"><title>Causal variants alter predicted transcription factor binding</title><p>Gene expression is shaped by chromatin state as well as by the binding of TFs to <italic>cis</italic>-regulatory elements (<xref ref-type="bibr" rid="bib87">Rando and Winston, 2012</xref>). Previous work suggested that while <italic>cis</italic>-eQTLs can perturb multiple aspects of chromatin, these molecular changes tend to ultimately be caused by sequence-directed differences in TF binding (<xref ref-type="bibr" rid="bib26">Degner et al., 2012</xref>; <xref ref-type="bibr" rid="bib47">Kasowski et al., 2013</xref>; <xref ref-type="bibr" rid="bib49">Kilpinen et al., 2013</xref>; <xref ref-type="bibr" rid="bib72">McVicker et al., 2013</xref>). To explore the influence of single causal variants on TF binding, we computed the expected propensity of each of 196 TFs to bind to the two alleles at each variant. Because some TFs have different affinities for the two DNA strands at their binding sites (<xref ref-type="bibr" rid="bib25">de Boer et al., 2020</xref>), we separately considered binding to the sense and the antisense strand, as well as in a strand-agnostic manner. TFs bind to individual ‘strong’ sites that closely match their motif, as well as to weaker sites with imperfect motif matches (<xref ref-type="bibr" rid="bib25">de Boer et al., 2020</xref>). To capture this distinction, we computed separate feature sets that considered only strong or only weak binding sites. These analyses cannot disambiguate sharing of similar binding motifs by different TFs but probe the overall role of perturbed TF binding in variant causality. Finally, we also aggregated the 196 TF-specific feature sets into summary features that reported maximum and average allelic differences at each variant across the 196 TFs (Materials and methods).</p><p>Overall, TF binding was strongly and broadly associated with variant causality (<xref ref-type="fig" rid="fig5">Figure 5B</xref>). Across the 2,940 TF-specific features, 481 features for 139 distinct TFs showed significant associations at an FDR of 5% (<xref ref-type="supplementary-material" rid="fig5sdata1">Figure 5—source data 1</xref>). All of the 27 summary features were significant at this threshold. For example, causal variants were more likely to result in gains or losses of strong binding sites (62%; n = 286) than non-causal variants (49%; n = 1,846; FET p=6e-8). These associations were found on both DNA strands (<xref ref-type="fig" rid="fig5">Figure 5B</xref>).</p><p>Causality was associated with differences in strong as well as weak TF binding (<xref ref-type="fig" rid="fig5">Figure 5B</xref>; <xref ref-type="fig" rid="fig5s1">Figure 5—figure supplement 1</xref>). While yeast promoters harbor relatively few strong binding sites, they contain many binding sites that are individually weak but that in combination greatly influence promoter activity (<xref ref-type="bibr" rid="bib25">de Boer et al., 2020</xref>). In our logistic regression analyses of the 196 individual TFs, differences in the various weak TF-binding metrics showed stronger associations with causality than those in the strong TF-binding metrics (<xref ref-type="fig" rid="fig5">Figure 5C</xref>). Because of the relatively low density of strong binding sites, many natural variants change no or only a small number of these sites. In contrast, the high density of weak binding sites presents a large mutational target, such that essentially every variant perturbs one or more weak binding sites. Evidently, these subtle differences can result in detectable expression changes. Overall, these analyses provided clear evidence that variants predicted to perturb TF binding are more likely to alter gene expression than variants not predicted to do so.</p></sec><sec id="s2-9"><title>Prediction of causal variants</title><p>The prediction of the effects of individual variants in a given genome is a major challenge for modern genetics and genomics. We tested whether the hundreds of causal variants we identified might enable us to predict the effects of individual <italic>cis</italic>-acting variants from their annotated features. We built 112 multiple logistic regression models for predicting variant causality from various feature subsets (<xref ref-type="supplementary-material" rid="fig6sdata1">Figure 6—source data 1</xref>). We trained these models using repeated 10-fold cross-validation on a random subset comprising 90% of the causal and non-causal variants and tested model performance on the remaining 10% of the variants. The best model achieved an area under the receiver operating curve (AUC) of 0.71 (<xref ref-type="fig" rid="fig6">Figure 6A</xref>; see inset for details on the model). Cross-validated elastic-net regularization, which is less prone to potential overfitting, did not improve model performance (best model AUC = 0.71).</p><fig id="fig6" position="float"><label>Figure 6.</label><caption><title>Prediction of causal variants and variant effects.</title><p>(<bold>A</bold>) Prediction results for 112 models. On the left, the plot shows the performance of binomial classifiers on the 10% test as black bars. On the right, the plot shows the performance of the linear predictors of variant effects as spearman correlation coefficients (rho) between actual and predicted log-fold changes in expression as pink bars. Blue symbols show r2 values for each model when fit to the entire dataset. The best classifier and linear model are indicated by orange bars and shown as insets. (<bold>B</bold>) Measured expression driven by each oligo versus expression predicted by the de Boer model. The dashed red line denotes the median measured expression level. (<bold>C</bold>) Observed fold-changes of individual variants measured in our MPRA versus fold-changes predicted by the de Boer model. The pink line shows the linear regression fit for all variants.</p><p><supplementary-material id="fig6sdata1"><label>Figure 6—source data 1.</label><caption><title>Multiple-feature regression analyses of causal variants.</title><p>Logistic and linear regression (to predict log-fold change) are in separate worksheets.</p></caption><media xlink:href="elife-62669-fig6-data1-v2.xlsx" mimetype="application" mime-subtype="xlsx"/></supplementary-material></p></caption><graphic xlink:href="elife-62669-fig6-v2.tif" mimetype="image" mime-subtype="tiff"/></fig><p>We also built a series of multiple linear regression models that used the same 112 feature combinations to predict the absolute log-fold change caused by a given variant. When fit to the entire dataset, nearly all models fit the data better than expected by chance, with r<sup>2</sup> values of up to 0.33 at a p-value of 3e-50 for a model including non-TF features as well as all TF features (<xref ref-type="fig" rid="fig6">Figure 6A</xref>). A model considering only the TF features fit the data almost as well (r<sup>2</sup> = 0.32, p=7e-49). The median r<sup>2</sup> across the 112 models was 0.09. Next, we used 90% of the variants to train these linear regression models using repeated 10-fold cross-validation and predicted absolute log-fold change in the remaining 10% of the variants. The best correlation between predicted and observed fold-changes was rho = 0.15 (p=0.0004). Several other models performed nearly as well (<xref ref-type="fig" rid="fig6">Figure 6A</xref>; <xref ref-type="supplementary-material" rid="fig6sdata1">Figure 6—source data 1</xref>). The median model prediction performance was rho = 0.07. Models fit with cross-validated elastic-net regularization performed no better than the corresponding unregularized linear models (best rho = 0.12, p=0.02; median rho = 0.005). Overall, prediction models built from various feature sets were able to predict the identity and effect size of causal variants better than chance, but with modest accuracy.</p><p>Recently, de Boer et al. reported a model for predicting reporter gene expression driven by millions of random DNA fragments (<xref ref-type="bibr" rid="bib25">de Boer et al., 2020</xref>). We tested whether this model was capable of predicting individual variant effects in our data. Using the DNA sequence centered on each variant in our library as input, we predicted reporter gene expression driven by each oligo in our libraries. These predicted expression values showed a significant correlation with measured expression in our MPRA (<xref ref-type="fig" rid="fig6">Figure 6B</xref>). We then computed the predicted fold change of each variant as the difference between the predicted expression for the respective BY and RM oligos. The predicted and measured fold-changes of all variants correlated significantly (<xref ref-type="fig" rid="fig6">Figure 6C</xref>). This correlation increased for variants that were causal in our data (<xref ref-type="fig" rid="fig6">Figure 6C</xref>), although this subset would be unknown during variant effect prediction from genome sequence alone. When we ignored the direction of the fold-changes, prediction performance was reduced for all variants (rho = 0.09, p=1e-8), and degraded completely for causal variants (rho = 0, p=0.9). This suggests that much of the ability of this independent model to predict variant effects in our data derives from the model’s ability to correctly capture whether altered binding of individual transcription factors to a given allele increases or decreases expression.</p><p>In summary, models based on individual features trained on our variant effects, as well as a completely independent model (<xref ref-type="bibr" rid="bib25">de Boer et al., 2020</xref>), showed similar ability to predict variant effects from DNA sequence. Although prediction performance was modest in both cases, the fact that a model trained on an independent experiment can capture some effects in our data demonstrates that the prediction of individual variant effects from DNA sequence alone is becoming possible.</p></sec></sec><sec sec-type="discussion" id="s3"><title>Discussion</title><p>In this work, we used an MPRA to examine the effects of roughly half of all intergenic DNA sequence variants between two yeast strains and identified hundreds of variants with significant, <italic>cis</italic>-acting effects on gene expression. These significant variants reflected the effects of known local eQTLs, suggesting that we have identified causal variants that underlie natural regulatory variation. Several insights emerged from in-depth analysis of these causal variants.</p><p>First, the molecular basis of local eQTLs can include single (as for <italic>OLE1</italic>) or multiple (as for <italic>CWP1</italic>) causal variants. Studies seeking to identify the molecular basis of QTL regions often implicitly assume that each QTL is driven by a single causal variant. However, efforts to fine-map QTLs for organismal traits often find that multiple, linked causal variants exist in a region (<xref ref-type="bibr" rid="bib58">Kroymann and Mitchell-Olds, 2005</xref>; <xref ref-type="bibr" rid="bib99">Sinha et al., 2008</xref>; <xref ref-type="bibr" rid="bib102">Steinmetz et al., 2002</xref>). Whether multiple variants also underlie QTLs for molecular traits such as gene expression was less clear, especially for local eQTLs in yeast, given the small mutational target size of the compact regulatory regions. Our results show that the molecular basis of local regulatory variation can involve multiple causal variants even in a single pair of yeast strains.</p><p>In some promoter regions (as for <italic>UFO1</italic>), we detected variants with effects in opposing directions that cancelled each other out, resulting in no local eQTL signal, demonstrating that QTL mapping can miss linked causal variants of opposite effect. Similarly, recent work has shown that the number of <italic>trans</italic>-acting loci affecting gene expression throughout the genome may have been greatly underestimated due to linkage in experimental crosses with limited recombination (<xref ref-type="bibr" rid="bib74">Metzger and Wittkopp, 2019</xref>). These results paint an emerging picture of a highly polygenic architecture of gene expression variation, with dozens of <italic>trans</italic>-acting loci across the genome (<xref ref-type="bibr" rid="bib4">Albert et al., 2018</xref>; <xref ref-type="bibr" rid="bib3">Albert et al., 2014b</xref>; <xref ref-type="bibr" rid="bib14">Brion et al., 2020</xref>; <xref ref-type="bibr" rid="bib74">Metzger and Wittkopp, 2019</xref>), each of which may harbor multiple causal variants.</p><p>Second, we detected several instances of non-additive epistatic interactions between variants in the same promoter. Previous work revealed epistatic eQTL pairs that influence the expression of a given gene, including between <italic>cis</italic> and <italic>trans</italic> eQTLs as well as between <italic>trans</italic> eQTLs (<xref ref-type="bibr" rid="bib4">Albert et al., 2018</xref>; <xref ref-type="bibr" rid="bib13">Brem et al., 2005</xref>). Very few of these cases have been resolved to individual variants (<xref ref-type="bibr" rid="bib13">Brem et al., 2005</xref>; <xref ref-type="bibr" rid="bib67">Lutz et al., 2019</xref>). Our results show that there can also be epistasis between multiple causal variants within a single local eQTL. Nevertheless, most of our assayed variant pairs act additively. Even for pairs with the most significant epistasis, the quantitative departures from additivity tended to be small (<xref ref-type="fig" rid="fig4">Figure 4</xref> &amp;S10). This result is consistent with searches for epistatic QTLs for various growth traits in this cross. Although examples of higher-order epistatic interactions with profound effects on some traits have been reported (<xref ref-type="bibr" rid="bib32">Forsberg et al., 2017</xref>), most of the hundreds of detected epistatic pairs involve small deviations from additivity that account for minuscule fractions of phenotypic variance (<xref ref-type="bibr" rid="bib4">Albert et al., 2018</xref>; <xref ref-type="bibr" rid="bib11">Bloom et al., 2015</xref>), as expected from quantitative genetic theory (<xref ref-type="bibr" rid="bib41">Hill et al., 2008</xref>).</p><p>Third, <italic>cis</italic>-acting variants showed signals of evolving under negative selection, in that the derived alleles of causal variants tended to have reduced population frequencies and were less likely to be found in the promoters of essential genes. Similar results reviewed in <xref ref-type="bibr" rid="bib97">Signor and Nuzhdin, 2018</xref> have been reported in yeast (<xref ref-type="bibr" rid="bib53">Kita et al., 2017</xref>; <xref ref-type="bibr" rid="bib90">Ronald and Akey, 2007</xref>), plants (<xref ref-type="bibr" rid="bib46">Josephs et al., 2015</xref>) and humans (<xref ref-type="bibr" rid="bib9">Battle et al., 2014</xref>), but were based on studies in crosses subject to linkage among neighboring variants, or in natural populations subject to both linkage disequilibrium and differences in statistical power for variants with different population frequencies. Our MPRA results are free from confounding effects of linkage and population frequency and extend the observation of negative selection to single, causal <italic>cis</italic>-acting variants.</p><p>Fourth, causal variants were enriched for sequence changes that altered predicted transcription factor binding. This enrichment was present for strong binding sites, as well as weak sites that do not pass strict binding site detection thresholds. Emerging evidence suggests that in addition to strong, high-affinity binding sites, intergenic DNA harbors an abundance of weak, low-affinity binding sites with imperfect motif matches (<xref ref-type="bibr" rid="bib25">de Boer et al., 2020</xref>; <xref ref-type="bibr" rid="bib106">Tanay, 2006</xref>; <xref ref-type="bibr" rid="bib118">Wunderlich and Mirny, 2009</xref>). This latter group provides a larger mutational target than the few strong exact motif matches. As a consequence, almost every natural variant may perturb one or multiple weak binding sites. The joint effects of perturbations to weak sites could further contribute to molecular architectures in which several causal variants shape the overall activity of a given promoter in a given individual. While more work is needed to explore the consequences of alterations to weak binding sites, they may also help explain why many inferred causal variants in human diseases are located close to, but do not obviously change, recognizable TF motifs (<xref ref-type="bibr" rid="bib31">Farh et al., 2015</xref>).</p><p>Our approach had several limitations. First, not all potential <italic>cis</italic>-acting variants were assayed in our libraries, including roughly half of intergenic variants as well as variants in gene regions such as the most proximal bases of the core promoter, the gene body, the 3’UTR, and the terminator. Future work with dedicated libraries to study these variants will likely reveal many additional causal effects. Meanwhile, the concordance of our results with local eQTLs suggests that we have discovered a non-trivial fraction of causal variants, and that, by extension, many <italic>cis</italic>-acting variants do indeed reside in promoter regions. Second, causal variants typically have small effects on gene expression, necessitating replicate assays to ensure high statistical power. Higher levels of replication, as well as further optimization of assay design, are likely to improve variant detection and effect quantification. Finally, our plasmid-bound assay cannot capture the full chromatin state of variants in the genome. Nevertheless, our causal variants correlated with local eQTL effects measured in the native genomic context, showing that chromatin state does not override the effects of all variants. This agreement was stronger for variants that reside in nucleosome-free regions than for nucleosome-bound variants. If nucleosomes do not form on the plasmids, such regulatory variants may be exposed to a regulatory environment similar to that in the genome. More broadly, the precise chromatin context in which a variant is embedded likely influences its exact effect on gene expression, consistent with the considerable differences revealed by MPRAs inserted at different sites in the human genome (<xref ref-type="bibr" rid="bib69">Maricque et al., 2019</xref>), as well as with the high degree of tissue-specificity of human <italic>cis</italic>-eQTLs (<xref ref-type="bibr" rid="bib36">GTEx Consortium et al., 2017</xref>).</p><p>The prediction of the consequences of individual DNA variants is a major area of research, especially given that accurate predictions could aid discovery of disease-causing mechanisms and diagnosis of individual patients with unknown genetic disorders. As a consequence, a plethora of computational methods aiming to predict the severity or molecular consequences of individual variants has been developed (for example, <xref ref-type="bibr" rid="bib42">Huang et al., 2017</xref>; <xref ref-type="bibr" rid="bib51">Kircher et al., 2014</xref>; <xref ref-type="bibr" rid="bib62">Lee et al., 2015</xref>; <xref ref-type="bibr" rid="bib122">Zhou et al., 2018</xref>; <xref ref-type="bibr" rid="bib123">Zhou and Troyanskaya, 2015</xref>, reviewed in <xref ref-type="bibr" rid="bib81">Nishizaki and Boyle, 2017</xref>). Although our models integrating various features to predict the identity and effect size of causal variants performed better than chance, no single feature and no simple combination of features achieved high prediction accuracy. While the size of our dataset, which included several hundred causal variants, may be insufficient to train predictors with better performance, an independent model trained on millions of DNA sequences (<xref ref-type="bibr" rid="bib25">de Boer et al., 2020</xref>) and applied to our data performed similarly. Our prediction accuracies are also comparable to those obtained from state-of-the-art machine-learning tools applied to human MPRA data (<xref ref-type="bibr" rid="bib52">Kircher et al., 2019</xref>) and to simulated genetic architectures (<xref ref-type="bibr" rid="bib65">Liu et al., 2019</xref>). Clearly, prediction of variant effects remains a challenging problem even in comparatively simple yeast promoters. Models trained in a range of environmental conditions, which would provide variation in chromatin contexts, as well as on MPRA data with longer and more diverse flanking sequences (<xref ref-type="bibr" rid="bib52">Kircher et al., 2019</xref>) may further improve predictive accuracy.</p></sec><sec sec-type="materials|methods" id="s4"><title>Materials and methods</title><table-wrap id="keyresource" position="anchor"><label>Key resources table</label><table frame="hsides" rules="groups"><thead><tr><th valign="top">Reagent type <break/>(species) or resource</th><th valign="top">Designation</th><th valign="top">Source or reference</th><th valign="top">Identifiers</th><th valign="top">Additional information</th></tr></thead><tbody><tr><td valign="top">Strain, strain background (<italic>Saccharomyces cerevisiae</italic>)</td><td valign="top">BY4741</td><td valign="top"><xref ref-type="bibr" rid="bib4">Albert et al., 2018</xref> (doi:<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.7554/eLife.35471">10.7554/eLife.35471</ext-link>)</td><td valign="top"/><td valign="top"/></tr><tr><td valign="top">Recombinant DNA reagent</td><td valign="top">RCP83 plasmid</td><td valign="top">This paper; (Addgene plasmid #163466)</td><td valign="top"/><td valign="top">Plasmid backbone</td></tr><tr><td valign="top">Recombinant DNA reagent</td><td valign="top">SurePrint Oligonucleotide Libraries</td><td valign="top">Agilent</td><td valign="top"/><td valign="top">Custom DNA oligo library</td></tr><tr><td valign="top">Gene (<italic>Aequorea victoria</italic>)</td><td valign="top">yEGFP</td><td valign="top">pKT0127 (Addgene plasmid #8728)</td><td valign="top"/><td valign="top">Reporter gene</td></tr><tr><td valign="top">Software, algorithm</td><td valign="top">R version 3.5.0</td><td valign="top"><ext-link ext-link-type="uri" xlink:href="https://www.r-project.org">https://www.r-project.org</ext-link></td><td valign="top"/><td valign="top">Data analysis</td></tr></tbody></table></table-wrap><p>MPRA design and analysis code generated for this paper is available at <ext-link ext-link-type="uri" xlink:href="https://github.com/frankwalbert/promoterVariants">https://github.com/frankwalbert/promoterVariants </ext-link>(copy archived at <ext-link ext-link-type="uri" xlink:href="https://archive.softwareheritage.org/swh:1:dir:e9626267bba430e5ba9d045629260763ff262441;origin=https://github.com/frankwalbert/promoterVariants;visit=swh:1:snp:3b5aa5fe84982530521c07efc43f79ef4b4fb634;anchor=swh:1:rev:fb7e232981f63281d944ccf273fdafa24ac2272d/">swh:1:rev:fb7e232981f63281d944ccf273fdafa24ac2272d</ext-link>; <xref ref-type="bibr" rid="bib5">Albert, 2020</xref>). Unless otherwise specified, analyses were performed in R (<ext-link ext-link-type="uri" xlink:href="https://www.r-project.org">https://www.r-project.org</ext-link>), using various packages from the tidyverse (<ext-link ext-link-type="uri" xlink:href="https://tidyverse.tidyverse.org/index.html">https://tidyverse.tidyverse.org/index.html</ext-link>) (<xref ref-type="bibr" rid="bib116">Wickham et al., 2019</xref>) and Bioconductor (<ext-link ext-link-type="uri" xlink:href="https://www.bioconductor.org">https://www.bioconductor.org</ext-link>) (<xref ref-type="bibr" rid="bib43">Huber et al., 2015</xref>). Specific packages are listed below. Sequences of primer and elements of the reporter construct are available in <xref ref-type="supplementary-material" rid="supp5">Supplementary file 5</xref>.</p><sec id="s4-1"><title>MPRA library design</title><p>Our design was based on 45,543 variants previously detected from short-read sequencing data of the BY and RM strains (<xref ref-type="bibr" rid="bib10">Bloom et al., 2013</xref>). Because the BY strain is nearly identical to the genome reference strain (<xref ref-type="bibr" rid="bib30">Engel et al., 2014</xref>), RM alleles are by definition ‘alternative’ alleles. Variants close to the telomeres were excluded as described in <xref ref-type="bibr" rid="bib3">Albert et al., 2014b</xref>. Variants included SNVs and short indels (<xref ref-type="supplementary-material" rid="supp6">Supplementary file 6</xref>). When multiple alternative alleles were reported for a variant, we used the alternative allele with the highest likelihood. We obtained gene annotations for the sacCer3 yeast reference genome build from SGD (<ext-link ext-link-type="uri" xlink:href="https://www.yeastgenome.org">https://www.yeastgenome.org</ext-link>) (<xref ref-type="bibr" rid="bib18">Cherry et al., 2012</xref>). Coding genes, as well as tRNAs, snoRNAs, snRNAs, and other non-coding RNAs were included in the design. Noncoding genes that overlapped a coding gene were removed, along with genes on the mitochondrial and 2µ plasmid genomes.</p><p>We defined transcription start sites based on full-length yeast transcript isoforms obtained by capturing and sequencing both the 5’ cap and 3’ polyadenylation site of yeast RNAs (<xref ref-type="bibr" rid="bib84">Pelechano et al., 2013</xref>). Specifically, we considered transcript isoforms reported in Supplementary Data S2 from <xref ref-type="bibr" rid="bib84">Pelechano et al., 2013</xref> that completely contained their annotated gene, avoiding isoforms that initiated transcription inside of annotated gene models or that terminated transcription prematurely. For each gene, we summed the number of reported counts for all isoforms that started at the same 5’ base in YPD medium. Based on these summed counts, we selected the 5’ position with the most read counts to represent the transcription start site for the given gene.</p><p>For the TSS library, we included DNA variants between BY and RM located within the 144 bp upstream of the transcription start site for each gene. We designed ‘oligo blocks’, sets of oligos carrying allelic versions of these 144 bp sequences. We extracted the reference genome sequence for these regions to form oligos carrying BY alleles at all variants. For each variant in the 144 bp, we created one additional oligo carrying the RM allele at this variant. We also generated one oligo per block that carried RM alleles at all variants. The TSS library comprised such oligo blocks for all genes that harbored variants in the design space of this library.</p><p>For the Upstream library, we considered all variants located between 72 bp upstream of the start codon of each gene and the coding sequence of the next upstream gene, up to a maximum distance of 1 kb. For each such variant, we created an oligo block consisting of one oligo that carried the BY allele at the variant flanked by the reference genome sequence centered on the variant and one oligo that carried the same flanking sequence but with the RM allele at the variant. At any other variants within the sequence covered by the oligo block, both oligos carried the BY allele such that each oligo block assayed a single variant. We subsampled 4547 variants from this design. Specifically, we included all variants located in promoters whose genes had at least some evidence (nominal p&lt;0.05) of allele-specific expression (ASE) in either of two ASE datasets (<xref ref-type="bibr" rid="bib4">Albert et al., 2018</xref>; <xref ref-type="bibr" rid="bib2">Albert et al., 2014a</xref>). We also included variants from 1000 randomly selected genes without evidence of ASE (p&gt;0.2), as well as 2062 additional variants selected at random from the remaining variants.</p><p>Both designs were strand-specific, such that for genes located on the plus strand we designed oligos based on the reference genome sequence, while for genes located on the minus strand, we used the reverse-complement of the reference sequence. Thus, in the finished reporter constructs, all oligos were correctly oriented with respect to the fluorescent reporter gene.</p><p>We removed oligo blocks in which at least one oligo happened to contain a restriction site that would be recognized by the restriction enzymes we used during cloning. We also removed oligo blocks in which RM insertion variants increased the length of at least one oligo to more than the synthesis limit of 200 bp.</p><p>In both libraries, we included an identical set of 200 oligos based on promoter fragments studied by <xref ref-type="bibr" rid="bib94">Sharon et al., 2012</xref>. Because the promoter fragments in that study were shorter (103 bp) than ours (144 bp), we added a random DNA sequence (<named-content content-type="sequence">TATAGAACGGAATCACCTCTGACAAGTAGCGTCAAATCGGT</named-content>) between the AscI cloning site and the p2 priming site such that this random sequence was removed during cloning (<xref ref-type="fig" rid="fig1s3">Figure 1—figure supplement 3</xref>). This extra sequence safeguarded against preferential PCR amplification of what would otherwise have been shorter oligo molecules.</p><p>To each designed oligo sequence, we added restriction sites and library-specific priming sites as shown in <xref ref-type="fig" rid="fig1s3">Figure 1—figure supplement 3</xref>. The resulting oligos had a median length of 195 bp. RM insertion variants increased the length of some oligos, such that the maximum oligo length was 200 bp. Because oligo synthesis is challenging for sequences rich in adenine (which are enriched in promoter sequences), we reverse-complemented the designed sequences of oligos that carried more adenines than thymines. Because the single-stranded oligo libraries are amplified by PCR prior to use, these reverse-complement oligos behaved equivalently equivalent to the designed oligos in downstream experiments. Oligos were purchased as one pool from Agilent Technologies.</p></sec><sec id="s4-2"><title>Reporter design</title><p>Inspired by the design of prior work (<xref ref-type="bibr" rid="bib94">Sharon et al., 2012</xref>), we placed the promoter fragment library upstream of a fluorescent yEGFP gene on a plasmid (<xref ref-type="fig" rid="fig1s3">Figure 1—figure supplement 3</xref>). Similar to <xref ref-type="bibr" rid="bib55">Kosuri et al., 2013</xref>, we measured reporter gene expression by quantifying yEGFP mRNA instead of fluorescence signal. Plasmids carried a CEN/ARS origin of replication and an <italic>ADH1</italic> terminator downstream of the reporter constructs. We included random 20 bp barcodes downstream of the yEGFP gene, such that the barcodes were transcribed as part of the yEGFP mRNA. Prior to insertion of the yEGFP gene into plasmids, we used paired-end sequencing to generate a dictionary linking each random barcode to the promoter fragment it tagged. We amplified the barcode from plasmid DNA or yEGFP cDNA to generate Illumina sequencing libraries. Reporter expression was quantified by high-throughput sequencing and counting of barcode reads. Details on these procedures are given below.</p><p>For the TSS library, the designed promoter fragments were inserted immediately upstream of the yEGFP ORF. For the Upstream library, an invariant <italic>HIS3</italic> minimal promoter was inserted between the oligo sequences and the yEGFP ORF (<xref ref-type="bibr" rid="bib94">Sharon et al., 2012</xref>).</p></sec><sec id="s4-3"><title>Initial library amplification and barcoding</title><p>The TSS and Upstream libraries were selectively amplified from the common oligo pool (Agilent). Per library, each of eight parallel reactions contained 0.2 µL of oligo pool diluted 1:10 in TE buffer for ~10 ng template DNA, 5 µL of 2x Kapa library amplification mix (Roche, KK2702), 0.2 µL forward primer (10 µM; TSS: orc281, Upstream: orc279), 0.2 µL reverse primer (10 µM; TSS: orc285; Upstream: orc283), and 4.4 µL water. Cycling conditions were: 98 °C for 2 min, 5 cycles of 98 °C for 3 s, 50 °C for 20 s, 60 °C for 10 s, 10 cycles of 98 °C for 3 s, 60 °C for 30 s, and a final extension at 60 °C for 2 min followed by a hold at 10 °C. The eight parallel reactions were pooled, cleaned up using QiaQuick PCR purification kits (Qiagen #28106), eluted in 30 µL EB buffer, and quantified with Qubit HS DNA kits (Thermo Fisher #Q32854).</p><p>To generate barcoded libraries, we performed PCR in eight parallel reactions. Each contained 10 ng of amplified library, 10 µL 5x Phusion buffer, 1 µL dNTP mix (10 mM, Invitrogen #18427013), 1 µL forward primer (10 µM; TSS: orc291, Upstream: orc287), 1 µL reverse primer (10 µM; TSS: orc292, Upstream: orc288), 0.5 µL Phusion HS II DNA Polymerase (ThermoFisher #F549L), and water to 50 µL. Cycling was performed as follows: 98 °C for 2 min, 14 cycles (TSS) or 12 cycles (Upstream) of 98 °C for 3 s, 60 °C for 30 s, 72 °C for 30 s, a final extension at 72 °C for 2 min, and a hold at 10 °C. The eight parallel reactions per library were combined, cleaned up using QiaQuick PCR purification kits, eluted in 30 µL EB buffer, and quantified with Qubit HS DNA kits.</p><p>Because the barcodes are part of the reporter mRNA molecules, their sequence potentially influences mRNA abundance in <italic>cis</italic>, for example by affecting mRNA stability. Indeed, pilot experiments suggested that barcodes with a high number of guanines had lower expression, in line with known effects of guanines in the 3’UTR (<xref ref-type="bibr" rid="bib93">Shalem et al., 2015</xref>). Barcodes starting with two guanines had particularly low expression, and an adenine at the first position increased expression variance. To increase overall library expression while maintaining high barcode complexity, all experiments reported here used random barcodes in which (1) the first base was either cytosine or thymine, (2) the next three positions and every remaining even position contained no guanines but each of the other three DNA bases at equal frequency, and (3) the remaining odd positions carried each of the four bases at equal frequency.</p></sec><sec id="s4-4"><title>Plasmid library generation</title><p>The base plasmid ‘RCP83’ was generated synthetically (Gen9/Ginkgo Bioworks) with a bacterial ampicillin marker and a yeast kanMX resistance cassette (<xref ref-type="supplementary-material" rid="supp7">Supplementary file 7</xref> contains the map of this plasmid after completed library construction; the RCP83 backbone is available at Addgene under ID #163466). The barcoded library was directionally ligated into the plasmid at an SfiI site. Upstream of this site, the plasmid contained the ‘PS1’ priming site used in barcode annotation (see below). Downstream of the SfiI site, the plasmid carried an <italic>ADH1</italic> terminator, ensuring efficient termination of the reporter constructs. We isolated RCP83 from <italic>E. coli</italic> cultures by maxi-prep using Qiagen Plasmid Plus Purification kits (Qiagen #12963, #12965) and quantified plasmid DNA using Qubit dsDNA HS Assay Kit (Thermo Fisher #Q32854). We digested 10 µg of RCP83 using 10 µL SfiI enzyme (NEB, #R0123S) in 50 µL 10x NEB CutSmart buffer and water to 500 µL. This mixture was incubated at 50 °C for 4 hr and cooled to room temperature. To prevent religation of the backbone, we added 10 µL of Shrimp Alkaline Phosphatase (NEB #M0371L) and incubated at 37 °C for 1 hr. We purified the digested vector from a 0.8% agarose gel (1.2 g agarose Fisher #21-255-00GM) in 150 mL 1x TAE (Fisher #FERB49) using a QiaQuick Gel Extraction Kit (Qiagen #28706) and quantified using a Qubit dsDNA HS Assay Kit.</p><p>We digested 1 µg of barcoded library using 5 µL SfiI enzyme, 10 µL 10x NEB CutSmart buffer, and water to 100 µL at 50 °C for 2 hr. Digested insert was purified using QiaQuick PCR purification and eluted in 30 µL EB.</p><p>Ligation of the digested library into the digested backbone was performed by mixing 1 µg of vector, 100 ng of library, 80 µL of 10x T4 DNA ligase buffer, 4 µL of T4 DNA ligase (NEB #M0202M), and water to 800 µL. The mixture was incubated at 16 °C for 18 hr, 65 °C for 20 min, and held at 12 °C. The reaction was cleaned up with a DNA Clean and Concentrator kit (Zymo Research, #D4013) and eluted in 25 µL water.</p><p>Transformation into <italic>E. coli</italic> was performed by electroporation into <italic>E. cloni</italic> 10G SUPREME Electrocompetent Cells (Lucigen, #60080–2), in 11 parallel transformations (1 mL each) on a Bio-Rad MicroPulser (Bio-Rad, #165–2100). The parallel transformations were combined and mixed with a total of 9.5 mL of recovery medium. To estimate the number of transformants obtained, we plated a dilution series (10-fold dilution steps from 1:100 to 1:10<sup>6</sup>) on LB carbenicillin (100 µg / mL) plates, grew the plates overnight at 30 °C and counted colonies after 24 hr. We obtained an approximate of 25 million and 230 million transformants for the TSS and Upstream libraries, respectively. Negative controls, for which we had performed the ligation and transformation with digested backbone but no insert, yielded negligible numbers of colonies. The transformed cells were plated on 15 cm LB + carbenicillin plates (500 µL per plate), grown overnight at 30 °C, and scraped into a total of 25 mL LB medium. The optical density (OD<sub>600</sub>) of these cells suspensions was determined, and 2 OD<sub>600</sub> units (~1 billion cells) were grown overnight at 30 °C in 400 mL LB + carbenicillin. After 24 hr, plasmids were isolated from 200 mL of these cultures using Qiagen Plasmid Plus Purification kits (Qiagen #12963, #12965). The remaining 200 mL of the overnight cultures were spun down, resuspended in 6 mL 20% glycerol, and frozen at −80 °C in 1 mL ultra-concentrated aliquots.</p></sec><sec id="s4-5"><title>Barcode annotation</title><p>We annotated designed promoters to random barcodes by high-throughput sequencing. To generate the sequencing libraries, we performed four parallel PCR reactions each for the TSS and the Upstream library. In each reaction, we combined 800 ng of plasmid library with 25 µL 2x Kapa library amplification mix (Roche, #KK2702), 1 µL forward primer (10 µM; orc295), 1 µL reverse primer (10 µM; orc296), and water to 50 µL. PCR was performed as follows: 98 °C for 2 min, 10 cycles of 98 °C for 3 s, 72 °C for 30 s, and a final extension at 72 °C for 2 min with a hold at 12 °C. The four reactions were combined, and clean-up performed using 360 µL AmPure XP beads for 200 µL combined reactions. Beads were incubated with the library at room temperature for 5 min, separated on a magnetic rack, washed twice with 500 µL 80% EtOH, and eluted from the beads with 50 µL elution (TE) buffer for 5 min. Sequencing libraries were quantified using qPCR Library Quantification kits (Roche, #KK4824).</p><p>We sequenced these annotation libraries in paired-end configuration with 250 bp read length on two lanes of an Illumina HiSeq 2500 instrument. At this read length, both of the paired reads covered the barcode and the associated oligo entirely. This configuration aided in distinguishing sequencing errors, which are present on only one of the paired reads, from synthesis and PCR errors in the sequenced molecule and are therefore present in both paired reads. Paired reads were merged using PEAR (<xref ref-type="bibr" rid="bib120">Zhang et al., 2014</xref>) using parameters -v 75 m 250 j 8 -y 2G -c 40 -b 64. Between 75 and 85% of read pairs were merged successfully.</p><p>We retained merged reads that contained two invariant sequences expected to be present in well-formed sequencing reads: in between the oligo and the barcode (TSS: <named-content content-type="sequence">CCTGCAGGGGTTTAGCCGGCGTG</named-content>; Upstream: <named-content content-type="sequence">CCTGCAGGGTTCCGCAGCCACAC</named-content>; these correspond to the AscI, p2, and SbfI sequences; <xref ref-type="fig" rid="fig1s3">Figure 1—figure supplement 3</xref>, <xref ref-type="supplementary-material" rid="supp5">Supplementary file 5</xref>) and upstream of the oligo (<named-content content-type="sequence">GGCCGTAATGGCC</named-content> in both libraries; corresponding to the SfiI-A site included in the synthetic oligo sequences; <xref ref-type="fig" rid="fig1s3">Figure 1—figure supplement 3</xref>, <xref ref-type="supplementary-material" rid="supp5">Supplementary file 5</xref>). These invariant sequences were used to locate the positions of the barcode and oligo in each read. This filter retained ~ 170 million reads in the TSS library and ~125 million reads in the Upstream library, representing 24 million (TSS) and 46 million (Upstream) unique barcodes. Most of these unique barcodes were rare (<xref ref-type="fig" rid="fig1s4">Figure 1—figure supplement 4</xref>), as indicated by median read counts per barcode of two (TSS) and one (Upstream). We did not group barcodes with similar sequences at this stage.</p><p>We examined barcodes that appeared to tag multiple oligos and concluded that the overwhelming majority of barcodes tagged a single oligo. Specifically, for the overwhelming majority of cases apparent tagging of multiple oligos, a single ‘primary’ oligo dominated the barcode, usually making up at least 90% of oligo reads for the barcode. Examination of the ‘secondary’ oligos showed that these usually differed from the primary oligo by a single nucleotide, most often involving a skipped base. Because our paired-end sequencing strategy reduced sequencing errors, we deemed it likely that these secondary oligos arose as PCR errors during barcoding or Illumina sequencing library generation. In downstream analyses, we only considered the primary oligo as the oligo tagged by the given barcode.</p><p>We next mapped these primary oligos to our designed sequences. Synthesis and PCR errors result in oligos that do not match the designed sequences perfectly. Because our interest was in the effects of individual sequence variants that usually alter just a single nucleotide, errors in the oligo molecules run the risk of dominating the modest effects exerted by natural regulatory variants. Therefore, we retained only barcodes for which the primary oligo perfectly matched a designed oligo. In the TSS library, we retained 9.2 million barcodes (38% of unique barcodes), while in the Upstream library we retained 20 million barcodes (43%).</p><p>In the TSS library, these barcodes tagged 6565 oligos, which was 91% of all designed TSS oligos. In the Upstream library, the barcodes tagged 9646 oligos, representing 98% of the design. Examination of designed oligos that were not present in the cloned libraries revealed that oligos starting with a guanine had lower representation in the library (<xref ref-type="fig" rid="fig1s5">Figure 1—figure supplement 5</xref>), especially for oligos that started with two guanines. This technical bias was more apparent in the TSS library, in which a higher fraction of oligos started with disfavored nucleotides than in the Upstream library.</p><p>Raw data and barcode assignments to oligos are available under GEO accession GSE155944 (<ext-link ext-link-type="uri" xlink:href="https://www.ncbi.nlm.nih.gov/geo/query/acc.cgi?acc=GSE155944">https://www.ncbi.nlm.nih.gov/geo/query/acc.cgi?acc=GSE155944</ext-link>).</p></sec><sec id="s4-6"><title>Reporter subcloning</title><p>After plasmid library generation, the fluorescent reporter gene was subcloned in between the promoter fragments and the barcodes of the annotated plasmid libraries by restriction digestion and ligation. Specifically, we used a codon-optimized yEGFP based on sequence from plasmid pKT0127, a gift from Kurt Thorn (Addgene plasmid #8728; <ext-link ext-link-type="uri" xlink:href="https://www.addgene.org/8728/">https://www.addgene.org/8728/</ext-link>). In the Upstream library, the 100 bases upstream of the <italic>HIS3</italic> gene (<xref ref-type="supplementary-material" rid="supp5">Supplementary file 5</xref>) were included as an invariant minimal promoter between the oligo library and yEGFP, as in <xref ref-type="bibr" rid="bib94">Sharon et al., 2012</xref>. yEGFP and the <italic>HIS3</italic> minimal promoter sequences were synthesized by Genewiz and were amplified using primers orc362_yeGFP (TSS) or orc362_HIS3 (Upstream), and orc363. These sequences were subcloned into the base plasmid (RCP83) to generate the final reporter plasmids.</p><p>Reporter gene inserts and the barcode-mapped plasmid library were digested as follows. We mixed 20 µg of insert or 15 µg plasmid library with 5 µL AscI (NEB R0558L), 5 µL SbfI-HF (NEB R3642L), 50 µL 10x CutSmart NEB buffer, and water to 500 µL. These mixtures were incubated at room temperature for 2 hr (insert) or 4 hr (plasmid library). The library plasmids were cooled to room temperature. We added 10 µL rSAP (NEB #M0371L) and incubated at 37 °C for 1 hr to prevent plasmid re-ligation. Digests were gel purified on 2% (insert) agarose or 0.8% (plasmid library), and bands at 700 bp (insert) or 6 kb (plasmid library) cut and purified using QiaQuick Gel Extraction Kit (Qiagen #28706) with elution into 30 µL EB.</p><p>Ligation of the digested reporter gene into the digested library plasmids was performed by mixing 2 µg (TSS) or 1 µg (Upstream) of vector, 700 ng (TSS) or 233 ng (Upstream) of insert, 80 µL of 10x T4 DNA ligase buffer, 8 µL of T4 DNA ligase (NEB #M0202M), and water to 800 µL. The mixture was incubated at 16 °C for 18 hr, 65 °C for 20 min, and held at 12 °C. The reaction was cleaned up and eluted in 25 µL water using the Zymo DNA clean and concentrator kit.</p><p>The ligated reactions were transformed into <italic>E. coli</italic> as described above for first step cloning, with 11 parallel transformations per library. Yields were 33 million (TSS) and 1.9 million (Upstream) cells per mL. Per library, 10 mL transformation mix was plated on twenty 15 cm LB + carbenicillin plates, grown overnight at 30 °C, and harvested by scraping in 5 mL LB per plate. From the resulting dense cell mix, we grew 2 billion cells in 100 mL LB + carbenicillin medium at 30 °C for ~ 20 hr for plasmid isolation. We condensed 45 mL of cells into 4 mL aliquots of 30% glycerol.</p><p>For the TSS library, we grew one aliquot overnight in 400 mL LB + carbenicillin medium at 30 °C and performed three maxi preps and two mega preps (200 mL each) using Qiagen Plasmid Plus Purification kits (Qiagen #12963, #12965, #12981). For the Upstream library, we sent glycerol stocks for large-scale plasmid prep (Genewiz) yielding 491 µg of plasmid library.</p><p>Library integrity was confirmed by restriction analysis with AscI and SbfI-HF. We also created next-generation sequencing libraries targeting the barcodes as described below and sequenced them on an Illumina MiSeq instrument to confirm that the libraries had retained high barcode complexity after subcloning.</p></sec><sec id="s4-7"><title>Yeast strain and media</title><p>We used a prototrophic yeast laboratory strain BY (<italic>MAT<bold>a</bold></italic>) without resistance cassettes (YLK1879) (<xref ref-type="bibr" rid="bib10">Bloom et al., 2013</xref>).</p><p>YNB (MSG) + glucose + G418 medium was prepared as follows: dissolve 6.7 g YNB without amino acids and without NH<sub>4</sub>SO<sub>4</sub> and 1 g monosodium glutamate in 900 mL H<sub>2</sub>O, autoclave, let cool to room temperature, add 100 mL sterile glucose (20%), add 1 mL G418 (1,000X).</p><p>YPD medium was made by combining 5 g yeast extract (Fisher #212720), 10 g peptone (Fisher #211820), and 435 mL milliQ water, autoclaving, and adding 50 mL 20% glucose (Fisher #901521).</p></sec><sec id="s4-8"><title>Yeast transformation</title><p>The libraries were transformed into using the LiAc method (<xref ref-type="bibr" rid="bib33">Gietz and Schiestl, 2007</xref>). For the TSS library, we transformed 1.6 L of yeast growing in YPD at an OD<sub>600</sub> of 1.25 with 248 µg plasmid DNA. We split the 1.6 L culture in half, harvested by centrifugation at 3,000 g for 5 min at 20 °C, washed twice with water (once with half and once with 1/5 of the culture volume), and pelleted by centrifugation. Each of the two cell pellets was mixed with transformation mix (28.8 mL PEG 50% w/v, 4.32 mL LiAc 1M, 6 mL single-stranded carrier DNA at 2 mg / mL, water to 43.2 mL, as well as half the plasmid DNA) in a 50 mL conical tube. Prior to making the transformation mix, single-stranded carrier DNA was denatured for 5 min in a boiling water bath and chilled on watery ice. Transformation mix was kept on ice until use. Cells mixed with transformation mix were heat shocked at 42 °C for 60 min in a shaking water bath. To recover the cells, we poured the transformation mixture into 1 L YPD medium pre-warmed to 30 °C and shook them at 30 °C for 4 hr. Recovered cells were spun at 5,000 g for 5 min, the liquid decanted, and the pellet resuspended in 8 mL PBS buffer. We plated the mixture on fourteen 150 × 15 mm plates (Fisher #08-757-14; 500 µL YPD + G418 per plate) and grew the cells at 30 °C for 2 d. A dilution series plated in parallel indicated 1.16 million cells per 500 µL plated cells, for a total of 16.2 million yeast transformants.</p><p>Plates were scraped with 5 mL PBS per plate, the harvested cells combined, spun (3,000 rpm for 5 min), and resuspended in 30 mL PBS. This cell suspension had an OD<sub>600</sub> of 128, as estimated by measuring a dilution series. We inoculated 400 million cells (104 µL) in 250 mL YNB (MSG) + glucose + G418 medium, grew them at 30 °C for 26 hr, measured OD<sub>600</sub>, and stored aliquots (750 µL culture in 750 µL 40% glycerol) at −80 °C. Each aliquot contained ~ 60 million cells.</p><p>The Upstream library was processed similarly, with the following differences. Transformation was performed using 320 µg of plasmid DNA and resulted in a total of 67 million transformants. About 100 OD units were harvested from the plates, and 1 billion cells (133 µL) grown overnight. Stored aliquots from this culture contained 250 million cells in 500 µL.</p></sec><sec id="s4-9"><title>Yeast growth</title><p>Samples were processed using two different sets of protocols. Briefly, the initial six TSS replicates were processed at smaller scale and with two-step PCR reactions. Below, this protocol is called ‘Protocol 1’. The remaining TSS samples and all Upstream samples were processed at larger scale and using one-step PCR reactions (‘Protocol 2’). See <xref ref-type="supplementary-material" rid="table1sdata1">Table 1—source data 1</xref> for further details on each sample.</p><sec id="s4-9-1"><title>Protocol 1</title><p>Aliquots of the glycerol stocks prepared during yeast transformation were thawed, spun down, the liquid removed, and the cells resuspended in 1 mL of YNB (MSG) + glucose + G418 medium. Resuspended cells were added to 100 mL medium per replicate and grown at 30 °C to an OD<sub>600</sub> of 0.4–0.5. At this OD, we harvested 50 million cells for DNA prep and 100 million cells for RNA prep, assuming 30 million cells per 1 mL culture at OD<sub>600</sub> = 1. Cells were spun down (5 min at 5,000 g), supernatant removed, and pellets placed on ice and stored at −80 °C as quickly as possible.</p></sec><sec id="s4-9-2"><title>Protocol 2</title><p>Aliquots of 500 million cells were inoculated in 650 mL of growth medium and grown to OD<sub>600</sub> = 0.4. From each replicate, we collected five pellets each from 40 to 50 mL of culture (~500 million cells each) for RNA extraction, and three to four pellets of 1–2 g of yeast for DNA extraction.</p></sec></sec><sec id="s4-10"><title>DNA extraction</title><sec id="s4-10-1"><title>Protocol 1</title><p>DNA was extracted using the Qiaprep Spin Miniprep kit (Qiagen, #27106) combined with an in-house yeast lysis protocol. To each sample of 50–60 million cells, we added 10 U zymolyase in 800 µL Y1 buffer (dissolve 182.2 g Sorbitol (Sigma Aldrich, #S6021) in 600 mL water, add 200 mL 0.5M Na<sub>2</sub>EDTA, pH eight solution (Ambion, #AM9262), add 1 mL 14.3M β-mercaptoethanol (Sigma Aldrich, #M3148), adjust to 1 L with water) and lysed the cells at 30 °C for 30 min in a Thermomixer at 750 rpm. Cells stripped off their cell walls were spun for 10 min at 300 g, and the supernatant removed. The remaining steps followed the Qiagen protocol. DNA was quantified using Qubit dsDNA HS assays (ThermoFisher #Q32854).</p></sec><sec id="s4-10-2"><title>Protocol 2</title><p>DNA was extracted from 1 to 2 g of yeast using the Qiagen Plasmid Plus Midi kit, following a user-supplied Qiagen protocol (‘QP11.doc Aug-01’ available from Qiagen). We performed two 200 µL elutions. DNA was further purified and concentrated using bead clean-up (Kapa; #KK8001) and eluted in 50 µL TE buffer.</p></sec></sec><sec id="s4-11"><title>RNA extraction</title><sec id="s4-11-1"><title>Protocol 1</title><p>Total RNA was extracted from 100 million cell aliquots using the Zymo Research Fungal/Bacterial RNA Miniprep kit (ZR, #R2014) including DNase I treatment. Bead beating was performed on a mini bead beater (Biospec). mRNA was purified using the Dynabeads mRNA Purification kit (Ambion #61006) and eluted in 10 µL of water. Total RNA and mRNA were quantified using the Qubit RNA BR kit.</p></sec><sec id="s4-11-2"><title>Protocol 2</title><p>Total RNA was extracted using the Qiagen RNeasy Maxi Kit, with bead beating in 96-well format on the mini bead beater and on-column DNAse digestion. Total RNA was concentrated using Amicon Ultra-0.5 filter columns (Millipore #UFC503024).</p></sec></sec><sec id="s4-12"><title>DNA sequencing library construction</title><sec id="s4-12-1"><title>Protocol 1</title><p>We used PCR to amplify the barcodes from the extracted plasmids and to generate Illumina TruSeq-compatible sequencing libraries. The resulting libraries all carried the same i5 index (which we did not sequence) and sample-specific i7 indexes that allowed multiplexed sequencing. Library generation was performed as two PCRs (<xref ref-type="fig" rid="fig1s6">Figure 1—figure supplement 6</xref>). The first PCR used universal primers (without Illumina indexes) to amplify the barcodes. Each reaction contained 7.5–10 ng of plasmid DNA, 10 µL of Phusion polymerase buffer, 1 µL dNTPs (10 mM; Invitrogen #18427013), 0.25 µL primer ‘common_ORF_v4’ (100 µM), 0.25 µL primer ‘RT_PCR_R_long’ (100 µM), 0.5 µL Phusion HotStart II HF Polymerase, and water to 50 µL. Cycling conditions were 98 °C for 30 s, 15 cycles of 98 °C for 10 s, 55 °C for 30 s, 72 °C for 15 s, and a final extension at 72 °C for 5 min followed by a hold at 8 °C. Reactions were cleaned up using Qiagen PCR purification kits, eluted in 30 µL EB, and quantified with Qubit HS DNA kits.</p><p>The 2<sup>nd</sup> PCR added sample-specific indexed primers. We adjusted the product of the first PCR to 25 ng in 10 µL, and assembled the same reaction mix as for the first PCR but with different primers: a sample-specific ‘RT_PCR_D7xx_F’ primer and the common ‘Illumina_PCR_R’ primer. The reaction was performed as above, but for six cycles and quantified with Qubit HS DNA kits.</p></sec><sec id="s4-12-2"><title>Protocol 2</title><p>We performed two-step PCR as above, with the following changes. Instead of Phusion, we used KAPA amplification mix (KK2602) for a reaction containing 25 µL 2X KAPA mix, 5 µL DNA, 19.5 µL water, 0.25 µL ‘common_ORF_v2’ (100 µM), and 0.25 µL primer ‘RT_PCR_R_long’ (100 µM). Cycling conditions, cleanup, quantification and adjustment to 25 ng / µL were the same as in Protocol 1. For the 2<sup>nd</sup> PCR, we performed two reactions per sample, which were then combined. For Upstream samples, we performed eight parallel reactions, in order to use the entire sample. PCR conditions were the same as for the second step PCR in Protocol 1, but with 1 µL of DNA and 23.5 µL water. Before pooling, individual samples were quantified using qPCR Library Quantification (KAPA, #KK4824).</p><p>For the 2018 Upstream samples (<xref ref-type="supplementary-material" rid="table1sdata1">Table 1—source data 1</xref>), we performed single step PCR on the DNA samples using the same reaction mix as in the 1<sup>st</sup> step PCR above, but with primers ‘RT_PCR_F_longD7xx’ and ‘RT-PCR-R-long’. PCR conditions were as above for the 1<sup>st</sup> PCR, but with 15 cycles.</p></sec></sec><sec id="s4-13"><title>RNA-sequencing library construction</title><sec id="s4-13-1"><title>Protocol 1</title><p>First-strand cDNA was synthesized by priming off the PS3 site immediately downstream of the barcodes. We combined 500 ng of mRNA in 4 µL water, 1 µL primer ‘RT_PCR_R_long’ (20 µM), 1 µL dNTPs (10 mM; Invitrogen #18427013), and 8.5 µL water, melted mRNA structure by incubating at 65 °C for 5 min, and placed the mixture on ice. We added 4 µL 5X reverse transcriptase buffer, 0.5 µL RiboLock RNAse inhibitor (ThermoFisher EO0381), and 1 µL Maxima Reverse Transcriptase (ThermoFisher EP0742), and incubated at 30 °C for 30 min and 85 °C for 5 min. To aid downstream PCR, we dissolved the cDNA/RNA hybrid by adding 1 µL ribonuclease H (Invitrogen 18021–014) and incubating at 37 °C for 20 min.</p><p>For second-strand cDNA synthesis, we performed a single extension using a forward primer that added a handle to be used in PCR. The corresponding handle for the first cDNA strand had already been added as part of the RT_PCR_R_long primer (<xref ref-type="fig" rid="fig1s6">Figure 1—figure supplement 6</xref>). We combined 5 µL of cDNA, 10 µL of 5X Phusion buffer, 1 µL dNTPs (10 mM; Invitrogen #18427013), 2.5 µL primer ‘common_ORF_v4’ (20 µM), 0.5 µL Phusion HotStart II HF Polymerase (ThermoFisher #F549L), and 31 µL water. We ran a single extension using the following thermocycler program: 98 °C for 30 s, 98°C for 10 s, 55°C for 30 s, 72°C for 15 s, 72°C for 5 min, and a hold at 8°C.</p><p>When present in a PCR reaction, the primers used above in cDNA synthesis would be capable of amplifying plasmid DNA. To prevent this, we degraded unincorporated primers by adding 5 µL of exonuclease I (ThermoFisher #EN0581), and incubating at 37 °C for 15 min and at 85 °C for 15 min. This last step at 85 °C inactivated the exonuclease but not the Phusion enzyme that was still present in the mixture, such that we were able to use the same mixture for additional PCR. To amplify the library and add i7 index primers, we added 0.25 µL primer ‘Illumina_PCR_R’ (100 µM) and 0.25 µL of a sample-specific ‘RT_PCR_D7xx_F’ primer (100 µM), and amplified as follows: 98 °C for 30 s, 18 cycles of 98 °C for 10 s, 55 °C for 30 s, 72 °C for 15 s, and a final extension at 72 °C for 5 min followed by a hold at 8 °C. The products were quantified with Qubit HS DNA kits.</p></sec><sec id="s4-13-2"><title>Protocol 2</title><p>First-strand cDNA synthesis was performed as in Protocol 1, with the following changes. We used SuperScript IV Reverse Transcriptase (Thermo Fisher, #18090–200) and incubated at 50 °C for 1 hr, using 2 µM ‘RT_PCR_R_long’ primer. To eliminate total RNA and cDNA/RNA hybrids after reverse transcription, we treated each cDNA sample with 1 µL RNase A (DNase and protease free, Thermo Fisher, #12091–021) and 1 µL RNase H (Thermo Fisher, #18021–071).</p><p>PCR was performed in a single step, using 5 µL 2x KAPA mix (Roche, #KK2602), 10 µL cDNA, 1 µL water, 0.25 µL sample-specific primer ‘RT_PCR_F_long_D7xx’ (100 µM), 0.25 µL common primer ‘Illumina_PCR_R’ (100 µM). PCR was performed as in Protocol 1, but for 15 cycles. For all Upstream as well as the 2018 TSS replicates, we performed parallel PCR reactions to capture as many cDNA molecules as possible. Before pooling, individual samples were quantified using qPCR Library Quantification (Roche, #KK4824).</p></sec></sec><sec id="s4-14"><title>Pooling and sequencing</title><p>All PCR products from a given batch, in particular those made from DNA and cDNA from the same sample, were quantified (using Qubit or qPCR, see above) and pooled to equal molarity. Pooled libraries were gel extracted using QiaQuick Gel Extraction kits from 2% gels with 1% EtBr, run alongside a 50 bp DNA ladder. The libraries were visible in sharp bands narrowly centered on 198 bp. Excised pools were eluted in 30 µL EB buffer. Pooled libraries were quantified using qPCR Library Quantification (KAPA, #KK4824).</p><p>Sequencing was performed on Illumina HiSeq 2500 instruments in ‘rapid mode’ using 15–20% phiX spike-in, reading 50 bp single ends and the i7 index.</p></sec><sec id="s4-15"><title>Barcode counting</title><p>Sequences were demultiplexed using a custom php script that allowed up to one mismatch between the expected and the observed index. Indexes for each sample are available in <xref ref-type="table" rid="table1">Table 1</xref> – Source Data one and <xref ref-type="supplementary-material" rid="supp5">Supplementary file 5</xref>. In each sample, we counted the number of reads that contained a unique barcode. Only barcodes that had been previously observed in the annotation runs were retained. Barcode reads that did not perfectly match the annotation were excluded. While this also excluded reads with sequence errors that could in principle be added to the analyses, the resulting data loss was minor: across all DNA and RNA samples, a median of 85% (a range of 75–90%) of reads mapped to known barcodes. Barcode counts for technical replicates, in which multiple indexed libraries had been created from the same growth culture, were summed. As a metric for the expression driven by each promoter oligo, we considered the log<sub>2</sub> ratio of summed RNA counts of all barcodes assigned to a given oligo, divided by the summed DNA counts. We used these ratios for visualization and comparison to other data, but not in significance testing (see below).</p><p>For two out of twelve TSS samples and two out of six Upstream samples, DNA counts were unavailable. We replaced these DNA counts with those from one other sample from the same batch (<xref ref-type="table" rid="table1">Table 1</xref> – Source Data 1). In the Upstream samples, we tested three additional strategies for addressing missing DNA values: replacement with the sum of all Upstream DNA samples, replacement with the counts from the Upstream annotation run, or sample exclusion. These different treatments did influence how many variants reached statistical significance, but only very slightly altered the estimated variant effect sizes (fold-changes). Eliminating the two Upstream samples with missing DNA would have resulted in fewer significant variants than we chose to report here, presumably due to reduced power (<xref ref-type="bibr" rid="bib79">Myint et al., 2019</xref>). Among the options that retained all samples with RNA counts, our choice of using DNA counts from another sample from the same batch resulted in the fewest significant variant effects. Thus, our treatment of missing DNA was conservative, while still allowing us to use RNA data from all available samples.</p><p>To compare the expression driven by promoter fragments to native gene expression in the genome, we computed the average expression driven by all fragments (irrespective of which allele they carried) designed from the promoter of the given gene.</p><p>Raw data and barcode counts are available under GEO accession GSE155944 (<ext-link ext-link-type="uri" xlink:href="https://www.ncbi.nlm.nih.gov/geo/query/acc.cgi?acc=GSE155944">https://www.ncbi.nlm.nih.gov/geo/query/acc.cgi?acc=GSE155944</ext-link>).</p></sec><sec id="s4-16"><title>Statistical analyses of variant effects</title><p>Tests for differential allele activity were performed in the ‘mpra’ R package (<xref ref-type="bibr" rid="bib79">Myint et al., 2019</xref>). At each variant, we performed pairwise comparisons of the expression driven by the RM allele versus that driven by the BY allele. When there were multiple variants within a promoter fragment assayed by the TSS library, we computed one test per variant such that the oligo carrying the RM allele for one particular variant was compared to the oligo carrying BY alleles at all variants.</p><p>We analyzed only those oligos that had summed barcode counts larger than zero in every replicate in both DNA and RNA data. After this filter, 2427 unique variants in 1429 promoters remained in the TSS library. In the Upstream library, 4467 variants in 1824 promoters remained.</p><p>We used the ‘sum’ barcode aggregation option in mpra, as well as mpra’s normalization for library size. When a variant was assayed multiple times on a given strand in the two libraries, we used the smallest p-value and its associated fold change for downstream analyses. FDR was calculated by the mpra package (<xref ref-type="bibr" rid="bib79">Myint et al., 2019</xref>).</p><p>To estimate the fraction of variants that have effects irrespective of their individual significance, we computed the π<sub>1</sub> statistic (<xref ref-type="bibr" rid="bib104">Storey and Tibshirani, 2003</xref>). To do so, we used the qvalue package in R (<xref ref-type="bibr" rid="bib103">Storey et al., 2020</xref>) to obtain the π<sub>0</sub> statistic (the proportion of statistical tests that are truly null) from the distribution of p-values of individual variants and then calculated π<sub>1</sub> = 1 – π<sub>0</sub>.</p></sec><sec id="s4-17"><title>Comparison of variant effects and local eQTLs</title><p>Local eQTL effects were obtained from <xref ref-type="bibr" rid="bib4">Albert et al., 2018</xref>. Effect sizes were expressed as log-fold changes of the RM allele relative to the BY allele. Because eQTLs can span wide genomic intervals, effect sizes and LOD scores of local eQTLs can differ slightly depending on whether they are determined at the peak marker of the given local eQTL or at the location of the gene itself. To use a metric that can be applied to all genes with MPRA data irrespective of whether the gene had a significant local eQTL, we used effects and LOD scores at the gene location. Confidence intervals for Spearman’s rank correlation were computed using the DescTools package (<xref ref-type="bibr" rid="bib98">Signorell et al., 2020</xref>).</p><p>Data on allele-specific mRNA expression in the BY/RM hybrid were obtained from Source Data seven in <xref ref-type="bibr" rid="bib4">Albert et al., 2018</xref>, which comprised allele-specific expression data from two datasets (<xref ref-type="bibr" rid="bib4">Albert et al., 2018</xref>; <xref ref-type="bibr" rid="bib2">Albert et al., 2014a</xref>). We performed a Fisher’s exact test on whether genes with at least one causal MPRA variant were more likely to show genome-wide significant allele-specific expression in at least one of the two datasets in <xref ref-type="bibr" rid="bib4">Albert et al., 2018</xref>.</p></sec><sec id="s4-18"><title>Non-additive interactions</title><p>We tested for non-additive interactions among pairs of variants assayed in the TSS library. Specifically, we assayed 342 variant pairs in promoters for which exactly two variants were present in the promoter regions assayed by the TSS library. For these variant pairs, our design included four oligos that represented all possible combinations of the two alleles at the two given variants. We retained only promoters in which all four oligos had summed barcode counts larger than zero in DNA and RNA in every replicate.</p><p>To test for interactions, we used the mpra package (<xref ref-type="bibr" rid="bib79">Myint et al., 2019</xref>) to fit the following model:<disp-formula id="equ1"><mml:math id="m1"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>y</mml:mi><mml:mo>=</mml:mo><mml:msub><mml:mi>β</mml:mi><mml:mn>0</mml:mn></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mi>β</mml:mi><mml:mn>1</mml:mn></mml:msub><mml:msub><mml:mi>x</mml:mi><mml:mn>1</mml:mn></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mi>β</mml:mi><mml:mn>2</mml:mn></mml:msub><mml:msub><mml:mi>x</mml:mi><mml:mn>2</mml:mn></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mi>β</mml:mi><mml:mn>3</mml:mn></mml:msub><mml:msub><mml:mi>x</mml:mi><mml:mn>1</mml:mn></mml:msub><mml:msub><mml:mi>x</mml:mi><mml:mn>2</mml:mn></mml:msub><mml:mo>+</mml:mo><mml:mi>ϵ</mml:mi></mml:mrow></mml:mstyle></mml:math></disp-formula></p><p>In the model, <italic>y</italic> is the expression driven by an oligo, <italic>x<sub>1</sub></italic> and <italic>x<sub>2</sub></italic> are indicator variables relating an observation to the given genotype, <italic>β<sub>0</sub></italic> is the intercept, <italic>β<sub>1</sub></italic> and <italic>β<sub>2</sub></italic> are the additive effects of the two variants, <italic>β<sub>3</sub></italic> is the interaction effect that we sought to test, and <italic>ε</italic> is the residual error. We contrasted this model to a model without the <italic>β<sub>3</sub>x<sub>1</sub>x<sub>2</sub></italic> term.</p></sec><sec id="s4-19"><title>Linkage disequilibrium</title><p>Pairwise linkage disequilibrium was calculated based on genotype data from a worldwide panel of yeast isolates (<xref ref-type="bibr" rid="bib85">Peter et al., 2018</xref>). We computed the D’ and r<sup>2</sup> linkage disequilibrium statistics as implemented in the ‘genetics’ R package (<xref ref-type="bibr" rid="bib112">Warnes, 2019</xref>), using the two most frequent alleles at each marker.</p></sec><sec id="s4-20"><title>Variant annotation for non-TF features</title><p>We gathered 31 non-TF features to describe each variant (<xref ref-type="supplementary-material" rid="supp3">Supplementary file 3</xref>). Gene annotations were obtained from <xref ref-type="bibr" rid="bib4">Albert et al., 2018</xref>. Nucleosome scores of a given variant in the genome were based on nucleosome occupancies reported in Supplementary Table 2 of <xref ref-type="bibr" rid="bib15">Brogaard et al., 2012</xref>, by extending the positions of the reported nucleosome center by 72 nucleotides in both directions. Nucleosome-free regions were defined as those with no reported occupancy. The number of TATA box motifs were counted for each variant allele and its flanking sequence using the consensus TATA(A/T)A(A/T)(A/G) (<xref ref-type="bibr" rid="bib8">Basehoar et al., 2004</xref>). Similarly, start codons were defined as occurrences of ‘ATG’. Ancestral alleles were defined as those present in the Taiwanese soil strain EN14S01 (‘standardized name’: ‘AMH’) from the highly diverged clade 17, using the genotype data from <xref ref-type="bibr" rid="bib85">Peter et al., 2018</xref>. We did not assign ancestral status for variants at which EN14S01 was heterozygous. Allele frequencies were obtained from genotypes in <xref ref-type="bibr" rid="bib85">Peter et al., 2018</xref>. PhastCons scores, which quantify nucleotide conservation across seven <italic>Saccharomyces</italic> species (<xref ref-type="bibr" rid="bib96">Siepel et al., 2005</xref>), were obtained for the positions corresponding to our variants from the UCSC genome browser (<ext-link ext-link-type="uri" xlink:href="https://genome.ucsc.edu/index.html">https://genome.ucsc.edu/index.html</ext-link>) (<xref ref-type="bibr" rid="bib37">Haeussler et al., 2019</xref>).</p></sec><sec id="s4-21"><title>Quantifying variant effects on predicted transcription factor binding</title><p>We obtained position weight matrices (PWMs) for 196 TFs from the ScerTF database (<xref ref-type="bibr" rid="bib101">Spivak and Stormo, 2012</xref>). For each variant and TF, we computed how well sequences containing the variant matched the TF’s PWM. Higher ‘TFBS scores’ represent better matches between the sequence and the PWM, increasing the probability that the TF binds the given sequence.</p><p>For a given variant, we calculated TFBS scores in sliding windows of width equal to the length of the PWM, which we moved over the variant in one-base-pair increments. To compute the TFBS scores, we summed the position weights from the PWM for the bases dictated by the sequence of the given window. We computed these windowed TFBS scores for the BY and the RM allele. For each allele, we retained the best (i.e. the highest) and the mean TFBS score across windows. In the text, we report results based on the absolute difference between the best TFBS scores of the two alleles. Differences between the mean TFBS scores of the two alleles yielded similar results (<xref ref-type="supplementary-material" rid="fig5sdata1">Figure 5—source data 1</xref>).</p><p>For each TF, the ScerTF database provides sequence score cutoffs, above which the given sequence is considered to be a ‘strong’ match to the PWM. We deemed TFBS scores below these thresholds to be ‘weak’ TFBSs. We computed TFBS scores separately for strong and weak matches. For strong TFBSs, we considered only sequence windows in which the TFBS score exceeded the sequence score cutoff. We also computed the number of strong binding sites for each TF by counting the number of windows that exceeded the cutoff. For weak TFBSs, we only considered TFBS scores below the cutoffs for each TF.</p><p>Together, these analyses yielded five metrics for how a variant is predicted to perturb the binding of a given TF: the allelic difference in (1) the best and (2) the mean TFBS score across windows for strong TFBSs, (3) the number of strong binding sites, (4) the best and (5) the mean TFBS score across windows for weak TFBSs.</p><p>To consider strand-specificity in variant effects on TF-binding, these five allelic TFBS metrics were computed across three strand contexts, for a total of 15 features per TF: (1) the sense (or ‘plus’) strand, that is the 5’ to 3’ sequence of nucleotides upstream of the reporter gene; (2) the antisense (or ‘minus’) strand, by analyzing the reverse complement of the plus strand; (3) ‘strand-agnostic’, in which we computed the difference between the best and mean scores across windows irrespective of which strand these scores came from. Across the 196 TFs, these features comprised a total set of 2940 features.</p><p>To obtain the aggregated TF features that summarized the 196 TF-specific sets of features, we computed the following 27 features for each variant. In each of the three strand contexts, we counted the total number of strong TFBSs changed (three features). Separately for strong and weak binding, we also computed the maximum difference in best or mean TFBS score across all TFs (12 features), and the average difference in best or mean TFBS score across all TFs (12 features).</p><p>To annotate individual TFBSs in <xref ref-type="fig" rid="fig5s1">Figure 5—figure supplement 1</xref>, we extracted the genome reference sequence flanking the two variants shown in the figure, up to a number of bases corresponding to the longest TF motif annotated in the Yeastract database (<xref ref-type="bibr" rid="bib76">Monteiro et al., 2020</xref>), and then uploaded these sequences with either the BY or the RM allele to the Yeastract web server (<ext-link ext-link-type="uri" xlink:href="http://www.yeastract.com">http://www.yeastract.com</ext-link>).</p></sec><sec id="s4-22"><title>Logistic regression tests of single features</title><p>We used logistic regression to test if each feature influenced the classification of a variant as causal. Causal variants were defined as those variants detected at an FDR of 5% or better. We contrasted this set to a set of non-causal variants with an unadjusted p-value larger than 0.2. Variants in between these two categories were excluded from the analyses.</p><p>Features were transformed to Z-scores using the formula:<disp-formula id="equ2"><mml:math id="m2"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>Z</mml:mi><mml:mo>=</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mi>X</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>−</mml:mo><mml:mi>μ</mml:mi><mml:mi>i</mml:mi><mml:mo stretchy="false">)</mml:mo><mml:mrow><mml:mo>/</mml:mo></mml:mrow><mml:msub><mml:mi>σ</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow></mml:mstyle></mml:math></disp-formula></p><p>Here, <italic>X<sub>i</sub></italic> is the value of the feature for variant <italic>i</italic>, and <italic>µ<sub>i</sub></italic> and <italic>σ<sub>I</sub></italic> refer to the mean and standard deviation of the feature values across all variants, respectively.</p><p>We performed logistic regression for each feature using the model:<disp-formula id="equ3"><mml:math id="m3"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:msub><mml:mi>S</mml:mi><mml:mi>i</mml:mi></mml:msub><mml:mo>=</mml:mo><mml:msub><mml:mi>β</mml:mi><mml:mn>0</mml:mn></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mi>β</mml:mi><mml:mrow><mml:mi>l</mml:mi><mml:mi>i</mml:mi><mml:mi>b</mml:mi><mml:mi>r</mml:mi><mml:mi>a</mml:mi><mml:mi>r</mml:mi><mml:mi>y</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mi>x</mml:mi><mml:mrow><mml:mi>l</mml:mi><mml:mi>i</mml:mi><mml:mi>b</mml:mi><mml:mi>r</mml:mi><mml:mi>a</mml:mi><mml:mi>r</mml:mi><mml:mi>y</mml:mi></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mi>β</mml:mi><mml:mrow><mml:mi>e</mml:mi><mml:mi>x</mml:mi><mml:mi>p</mml:mi><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:mi>s</mml:mi><mml:mi>s</mml:mi><mml:mi>i</mml:mi><mml:mi>o</mml:mi><mml:mi>n</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mi>x</mml:mi><mml:mrow><mml:mi>e</mml:mi><mml:mi>x</mml:mi><mml:mi>p</mml:mi><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:mi>s</mml:mi><mml:mi>s</mml:mi><mml:mi>i</mml:mi><mml:mi>o</mml:mi><mml:mi>n</mml:mi></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:msub><mml:mi>β</mml:mi><mml:mrow><mml:mi>f</mml:mi><mml:mi>e</mml:mi><mml:mi>a</mml:mi><mml:mi>t</mml:mi><mml:mi>u</mml:mi><mml:mi>r</mml:mi><mml:mi>e</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mi>x</mml:mi><mml:mrow><mml:mi>f</mml:mi><mml:mi>e</mml:mi><mml:mi>a</mml:mi><mml:mi>t</mml:mi><mml:mi>u</mml:mi><mml:mi>r</mml:mi><mml:mi>e</mml:mi></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:mi>ϵ</mml:mi></mml:mrow></mml:mstyle></mml:math></disp-formula></p><p>Here, the response variable <italic>S<sub>i</sub></italic> is a binary vector indicating whether variant <italic>i</italic> is significant. We included the library a variant was measured in and the average normalized baseline expression driven by the oligos used to measure the variant as covariates in the models. <italic>β<sub>library</sub></italic> and <italic>β<sub>expression</sub></italic> are the effects of these two possible confounders, while <italic>x<sub>library</sub></italic> and <italic>x<sub>expression</sub></italic> are indicator variables relating an observation to the given covariate. <italic>β<sub>0</sub></italic> is the overall intercept, and <italic>ε</italic> is the residual error. <italic>β<sub>feature</sub></italic> is the effect of the given feature on the probability that a variant is significant, and <italic>x<sub>feature</sub></italic> is an indicator variable relating the feature values to the variants.</p><p>We used ANOVA to contrast this model to one without the <italic>β<sub>feature</sub>x<sub>feature</sub></italic> term. We corrected for multiple testing by computing FDR as q-values using the ‘qvalue’ package (<xref ref-type="bibr" rid="bib104">Storey and Tibshirani, 2003</xref>).</p></sec><sec id="s4-23"><title>Multiple regression models</title><p>To build predictors of variant causality (‘causal’ vs ‘non-causal’), we first divided our set of features into 112 partially redundant feature subsets. These 112 models differed in whether they considered the non-TF features, whether they included TF features from the plus, minus, and/or the strand-agnostic set, whether they considered strong and/or weak TF metrics, whether they included the aggregated TF summary features, and whether they included only features that had been significant in the single regression analyses. Details on each model are listed in <xref ref-type="supplementary-material" rid="fig6sdata1">Figure 6—source data 1</xref>.</p><p>For each of the 112 feature subsets, we built a logistic regression model to predict variant causality. For building these models, we first divided the variants also used in single-feature analyses into a training set comprising 90% of the data, and a test set comprising 10% of the data. Training was performed using repeated 10-fold cross validation as follows. The training set was split ten times, each time considering one of ten possible non-overlapping fractions of 10% of the training set as a validation set. In each of these ten splits, we used the remaining 90% of the training set to fit the model and computed Cohen’s Kappa on the given validation set. This process was repeated five times, on five different 10-fold splits of the training data. The models were trained to maximize Cohen’s Kappa on the validation sets. After training, each of the 112 models was applied to the 10% test set that had been held out from training. Model accuracy for the logistic regression models was measured by the area under the receiver operator characteristic curve (AUC). Elastic-net logistic regression was performed using the ‘caret’ package (<xref ref-type="bibr" rid="bib59">Kuhn et al., 2020</xref>). We excluded 35 features with an ‘NA’ value (27 aggregated TFBS features, expression level of the associated gene, two measures of dNdS, PPI, three measures of synthetic genetic interactions, nucleosome score).</p><p>Linear regression models to predict variant effects (expressed as absolute log-fold changes) were performed using the same repeated 10-fold cross-validation scheme and trained to minimize the root-mean squared error (RMSE). Prediction accuracy was measured using Spearman correlation coefficients between predicted and actual data on the test set. All model training was done using the ‘caret’ package (<xref ref-type="bibr" rid="bib59">Kuhn et al., 2020</xref>).</p><p>In the case of multiple linear regressions, we also calculated model fit as the R<sup>2</sup> on the entire set of variants used in the single-feature regressions. These linear models were fit using the ‘lm’ function in R.</p><p>Elastic-net regression models predicting variant effects followed the same splitting and training strategy as for the 112 non-regularized models. We tuned the model across 10 by 10 combinations of the mixing parameter (alpha) and the regularization parameter (lambda) to select the model with the lowest RMSE in cross-validation. The best model was used to predict on the 10% test set, and model accuracy was measured as above. Elastic nets were trained using the ‘caret’ package (<xref ref-type="bibr" rid="bib59">Kuhn et al., 2020</xref>).</p></sec><sec id="s4-24"><title>Predicting variant effects on our data using the <xref ref-type="bibr" rid="bib25">de Boer et al., 2020</xref> model</title><p>To predict gene expression corresponding to our variants, we used the non-positional YPD model with pTpA embedding from <xref ref-type="bibr" rid="bib25">de Boer et al., 2020</xref>. Among the models trained in that study, this model is least sensitive to the surrounding sequence context, which differs between our sequences and those used by de Boer et al. We used the model trained in YPD medium; the same medium we used in our experiments. The model uses 110 bp of promoter sequence as input to predict gene expression. Because our oligos were 144 bp long, we could not use the complete oligo sequence for prediction. Instead, we used 110 bp of DNA centered on each variant. For each variant, we used two 110 bp sequences, one containing the BY allele and the other containing the RM allele. The sequences were one-hot encoded using code from de Boer et al. and then fed into the model.</p><p>The predicted expression values for the BY and RM alleles of each variant were subtracted to predict fold change between alleles. Predicted expression and variant effects were compared to those observed in our experiments. As observed oligo expression values, we used the ‘AveExpr’ value in <xref ref-type="supplementary-material" rid="supp3">Supplementary file 3</xref> as the expression of the BY allele, and the sum of the ‘AveExpr’ and ‘logFC’ as the expression of the RM allele. The ‘logFC’ was used as the observed variant effect.</p></sec></sec></body><back><ack id="ack"><title>Acknowledgements</title><p>We are grateful to Joshua Bloom, Chad Myers, and Henry Ward for input on data analysis. We thank Carl de Boer for help with applying his gene expression prediction model, Joshua Bloom for providing the BY/RM genotype data, and Liangke Gou for help with nucleosome data. We thank Suhua Feng for assistance with Illumina sequencing. We acknowledge resources and support from the Minnesota Supercomputing Institute.</p></ack><sec sec-type="additional-information" id="s5"><title>Additional information</title><fn-group content-type="competing-interest"><title>Competing interests</title><fn fn-type="COI-statement" id="conf1"><p>No competing interests declared</p></fn></fn-group><fn-group content-type="author-contribution"><title>Author contributions</title><fn fn-type="con" id="con1"><p>Data curation, Software, Formal analysis, Investigation, Visualization, Methodology, Writing - original draft, Writing - review and editing</p></fn><fn fn-type="con" id="con2"><p>Conceptualization, Investigation, Methodology, Writing - review and editing</p></fn><fn fn-type="con" id="con3"><p>Data curation, Investigation, Methodology</p></fn><fn fn-type="con" id="con4"><p>Conceptualization, Resources, Supervision, Funding acquisition, Methodology, Writing - review and editing</p></fn><fn fn-type="con" id="con5"><p>Conceptualization, Resources, Supervision, Funding acquisition, Writing - review and editing</p></fn><fn fn-type="con" id="con6"><p>Conceptualization, Data curation, Software, Formal analysis, Supervision, Funding acquisition, Validation, Investigation, Visualization, Methodology, Writing - original draft, Project administration, Writing - review and editing</p></fn></fn-group></sec><sec sec-type="supplementary-material" id="s6"><title>Additional files</title><supplementary-material id="supp1"><label>Supplementary file 1.</label><caption><title>Oligo design.</title></caption><media xlink:href="elife-62669-supp1-v2.xlsx" mimetype="application" mime-subtype="xlsx"/></supplementary-material><supplementary-material id="supp2"><label>Supplementary file 2.</label><caption><title>Oligo counts per replicate.</title></caption><media xlink:href="elife-62669-supp2-v2.xls" mimetype="application" mime-subtype="excel"/></supplementary-material><supplementary-material id="supp3"><label>Supplementary file 3.</label><caption><title>Non-TF features for each variant, along with statistical results for each variant.</title></caption><media xlink:href="elife-62669-supp3-v2.xlsx" mimetype="application" mime-subtype="xlsx"/></supplementary-material><supplementary-material id="supp4"><label>Supplementary file 4.</label><caption><title>All features used in variant annotation, including TFBS.</title><p>(gzipped text file).</p></caption><media xlink:href="elife-62669-supp4-v2.txt.gz" mimetype="application" mime-subtype="x-gzip"/></supplementary-material><supplementary-material id="supp5"><label>Supplementary file 5.</label><caption><title>Primers sequences and sequences of various components of the reporter gene construct.</title></caption><media xlink:href="elife-62669-supp5-v2.xlsx" mimetype="application" mime-subtype="xlsx"/></supplementary-material><supplementary-material id="supp6"><label>Supplementary file 6.</label><caption><title>BY/RM sequence variants used in MPRA design (gzipped vcf file).</title></caption><media xlink:href="elife-62669-supp6-v2.vcf.gz" mimetype="application" mime-subtype="x-gzip"/></supplementary-material><supplementary-material id="supp7"><label>Supplementary file 7.</label><caption><title>Sequence and map of a plasmid in the library after completed library construction.</title><p>In place of the library, the sequence contains an example promoter fragment.</p></caption><media xlink:href="elife-62669-supp7-v2.gb.txt" mimetype="text" mime-subtype="plain"/></supplementary-material><supplementary-material id="transrepform"><label>Transparent reporting form</label><media xlink:href="elife-62669-transrepform-v2.docx" mimetype="application" mime-subtype="docx"/></supplementary-material></sec><sec sec-type="data-availability" id="s7"><title>Data availability</title><p>Raw data and barcode assignments to oligos are available under GEO accession GSE155944. Source Data is provided for Figures 2, 3, 4, 5, and 6. Additional processed data and the MPRA design are available as Supplementary Files.</p><p>The following dataset was generated:</p><p><element-citation publication-type="data" specific-use="isSupplementedBy" id="dataset1"><person-group person-group-type="author"><collab>Renganaath</collab><collab>Chong</collab></person-group><year iso-8601-date="2020">2020</year><data-title>Massively parallel identification of cis-regulatory variants in yeast promoters</data-title><source>NCBI Gene Expression Omnibus</source><pub-id assigning-authority="NCBI" pub-id-type="accession" xlink:href="https://www.ncbi.nlm.nih.gov/geo/query/acc.cgi?acc=GSE155944">GSE155944</pub-id></element-citation></p></sec><ref-list><title>References</title><ref id="bib1"><element-citation publication-type="journal"><person-group person-group-type="author"><collab>1000 Genomes Project Consortium</collab><name><surname>Auton</surname> <given-names>A</given-names></name><name><surname>Brooks</surname> <given-names>LD</given-names></name><name><surname>Durbin</surname> <given-names>RM</given-names></name><name><surname>Garrison</surname> <given-names>EP</given-names></name><name><surname>Kang</surname> <given-names>HM</given-names></name><name><surname>Korbel</surname> <given-names>JO</given-names></name><name><surname>Marchini</surname> <given-names>JL</given-names></name><name><surname>McCarthy</surname> <given-names>S</given-names></name><name><surname>McVean</surname> <given-names>GA</given-names></name><name><surname>Abecasis</surname> <given-names>GR</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>A global reference for human genetic variation</article-title><source>Nature</source><volume>526</volume><fpage>68</fpage><lpage>74</lpage><pub-id pub-id-type="doi">10.1038/nature15393</pub-id><pub-id pub-id-type="pmid">26432245</pub-id></element-citation></ref><ref id="bib2"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Albert</surname> <given-names>FW</given-names></name><name><surname>Muzzey</surname> <given-names>D</given-names></name><name><surname>Weissman</surname> <given-names>JS</given-names></name><name><surname>Kruglyak</surname> <given-names>L</given-names></name></person-group><year iso-8601-date="2014">2014a</year><article-title>Genetic influences on translation in yeast</article-title><source>PLOS Genetics</source><volume>10</volume><elocation-id>e1004692</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pgen.1004692</pub-id><pub-id pub-id-type="pmid">25340754</pub-id></element-citation></ref><ref id="bib3"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Albert</surname> <given-names>FW</given-names></name><name><surname>Treusch</surname> <given-names>S</given-names></name><name><surname>Shockley</surname> <given-names>AH</given-names></name><name><surname>Bloom</surname> <given-names>JS</given-names></name><name><surname>Kruglyak</surname> <given-names>L</given-names></name></person-group><year iso-8601-date="2014">2014b</year><article-title>Genetics of single-cell protein abundance variation in large yeast populations</article-title><source>Nature</source><volume>506</volume><fpage>494</fpage><lpage>497</lpage><pub-id pub-id-type="doi">10.1038/nature12904</pub-id><pub-id pub-id-type="pmid">24402228</pub-id></element-citation></ref><ref id="bib4"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Albert</surname> <given-names>FW</given-names></name><name><surname>Bloom</surname> <given-names>JS</given-names></name><name><surname>Siegel</surname> <given-names>J</given-names></name><name><surname>Day</surname> <given-names>L</given-names></name><name><surname>Kruglyak</surname> <given-names>L</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Genetics of <italic>trans</italic>-regulatory variation in gene expression</article-title><source>eLife</source><volume>7</volume><elocation-id>e35471</elocation-id><pub-id pub-id-type="doi">10.7554/eLife.35471</pub-id><pub-id pub-id-type="pmid">30014850</pub-id></element-citation></ref><ref id="bib5"><element-citation publication-type="software"><person-group person-group-type="author"><name><surname>Albert</surname> <given-names>FW</given-names></name></person-group><year iso-8601-date="2020">2020</year><data-title>promoterVariants</data-title><source>Software Heritage</source><version designator="swh:1:rev:fb7e232981f63281d944ccf273fdafa24ac2272d">swh:1:rev:fb7e232981f63281d944ccf273fdafa24ac2272d</version><ext-link ext-link-type="uri" xlink:href="https://archive.softwareheritage.org/swh:1:dir:e9626267bba430e5ba9d045629260763ff262441;origin=https://github.com/frankwalbert/promoterVariants;visit=swh:1:snp:3b5aa5fe84982530521c07efc43f79ef4b4fb634;anchor=swh:1:rev:fb7e232981f63281d944ccf273fdafa24ac2272d">https://archive.softwareheritage.org/swh:1:dir:e9626267bba430e5ba9d045629260763ff262441;origin=https://github.com/frankwalbert/promoterVariants;visit=swh:1:snp:3b5aa5fe84982530521c07efc43f79ef4b4fb634;anchor=swh:1:rev:fb7e232981f63281d944ccf273fdafa24ac2272d</ext-link></element-citation></ref><ref id="bib6"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Albert</surname> <given-names>FW</given-names></name><name><surname>Kruglyak</surname> <given-names>L</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>The role of regulatory variation in complex traits and disease</article-title><source>Nature Reviews Genetics</source><volume>16</volume><fpage>197</fpage><lpage>212</lpage><pub-id pub-id-type="doi">10.1038/nrg3891</pub-id><pub-id pub-id-type="pmid">25707927</pub-id></element-citation></ref><ref id="bib7"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Arnold</surname> <given-names>CD</given-names></name><name><surname>Gerlach</surname> <given-names>D</given-names></name><name><surname>Stelzer</surname> <given-names>C</given-names></name><name><surname>Boryń</surname> <given-names>ŁM</given-names></name><name><surname>Rath</surname> <given-names>M</given-names></name><name><surname>Stark</surname> <given-names>A</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>Genome-wide quantitative enhancer activity maps identified by STARR-seq</article-title><source>Science</source><volume>339</volume><fpage>1074</fpage><lpage>1077</lpage><pub-id pub-id-type="doi">10.1126/science.1232542</pub-id><pub-id pub-id-type="pmid">23328393</pub-id></element-citation></ref><ref id="bib8"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Basehoar</surname> <given-names>AD</given-names></name><name><surname>Zanton</surname> <given-names>SJ</given-names></name><name><surname>Pugh</surname> <given-names>BF</given-names></name></person-group><year iso-8601-date="2004">2004</year><article-title>Identification and distinct regulation of yeast TATA box-containing genes</article-title><source>Cell</source><volume>116</volume><fpage>699</fpage><lpage>709</lpage><pub-id pub-id-type="doi">10.1016/S0092-8674(04)00205-3</pub-id><pub-id pub-id-type="pmid">15006352</pub-id></element-citation></ref><ref id="bib9"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Battle</surname> <given-names>A</given-names></name><name><surname>Mostafavi</surname> <given-names>S</given-names></name><name><surname>Zhu</surname> <given-names>X</given-names></name><name><surname>Potash</surname> <given-names>JB</given-names></name><name><surname>Weissman</surname> <given-names>MM</given-names></name><name><surname>McCormick</surname> <given-names>C</given-names></name><name><surname>Haudenschild</surname> <given-names>CD</given-names></name><name><surname>Beckman</surname> <given-names>KB</given-names></name><name><surname>Shi</surname> <given-names>J</given-names></name><name><surname>Mei</surname> <given-names>R</given-names></name><name><surname>Urban</surname> <given-names>AE</given-names></name><name><surname>Montgomery</surname> <given-names>SB</given-names></name><name><surname>Levinson</surname> <given-names>DF</given-names></name><name><surname>Koller</surname> <given-names>D</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>Characterizing the genetic basis of transcriptome diversity through RNA-sequencing of 922 individuals</article-title><source>Genome Research</source><volume>24</volume><fpage>14</fpage><lpage>24</lpage><pub-id pub-id-type="doi">10.1101/gr.155192.113</pub-id><pub-id pub-id-type="pmid">24092820</pub-id></element-citation></ref><ref id="bib10"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Bloom</surname> <given-names>JS</given-names></name><name><surname>Ehrenreich</surname> <given-names>IM</given-names></name><name><surname>Loo</surname> <given-names>WT</given-names></name><name><surname>Lite</surname> <given-names>TL</given-names></name><name><surname>Kruglyak</surname> <given-names>L</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>Finding the sources of missing heritability in a yeast cross</article-title><source>Nature</source><volume>494</volume><fpage>234</fpage><lpage>237</lpage><pub-id pub-id-type="doi">10.1038/nature11867</pub-id><pub-id pub-id-type="pmid">23376951</pub-id></element-citation></ref><ref id="bib11"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Bloom</surname> <given-names>JS</given-names></name><name><surname>Kotenko</surname> <given-names>I</given-names></name><name><surname>Sadhu</surname> <given-names>MJ</given-names></name><name><surname>Treusch</surname> <given-names>S</given-names></name><name><surname>Albert</surname> <given-names>FW</given-names></name><name><surname>Kruglyak</surname> <given-names>L</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>Genetic interactions contribute less than additive effects to quantitative trait variation in yeast</article-title><source>Nature Communications</source><volume>6</volume><elocation-id>8712</elocation-id><pub-id pub-id-type="doi">10.1038/ncomms9712</pub-id><pub-id pub-id-type="pmid">26537231</pub-id></element-citation></ref><ref id="bib12"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Brem</surname> <given-names>RB</given-names></name><name><surname>Yvert</surname> <given-names>G</given-names></name><name><surname>Clinton</surname> <given-names>R</given-names></name><name><surname>Kruglyak</surname> <given-names>L</given-names></name></person-group><year iso-8601-date="2002">2002</year><article-title>Genetic dissection of transcriptional regulation in budding yeast</article-title><source>Science</source><volume>296</volume><fpage>752</fpage><lpage>755</lpage><pub-id pub-id-type="doi">10.1126/science.1069516</pub-id><pub-id pub-id-type="pmid">11923494</pub-id></element-citation></ref><ref id="bib13"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Brem</surname> <given-names>RB</given-names></name><name><surname>Storey</surname> <given-names>JD</given-names></name><name><surname>Whittle</surname> <given-names>J</given-names></name><name><surname>Kruglyak</surname> <given-names>L</given-names></name></person-group><year iso-8601-date="2005">2005</year><article-title>Genetic interactions between polymorphisms that affect gene expression in yeast</article-title><source>Nature</source><volume>436</volume><fpage>701</fpage><lpage>703</lpage><pub-id pub-id-type="doi">10.1038/nature03865</pub-id></element-citation></ref><ref id="bib14"><element-citation publication-type="preprint"><person-group person-group-type="author"><name><surname>Brion</surname> <given-names>C</given-names></name><name><surname>Lutz</surname> <given-names>S</given-names></name><name><surname>Albert</surname> <given-names>FW</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>Simultaneous quantification of mRNA and protein in single cells reveals post-transcriptional effects of genetic variation</article-title><source>bioRxiv</source><pub-id pub-id-type="doi">10.1101/2020.07.02.185413</pub-id></element-citation></ref><ref id="bib15"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Brogaard</surname> <given-names>K</given-names></name><name><surname>Xi</surname> <given-names>L</given-names></name><name><surname>Wang</surname> <given-names>JP</given-names></name><name><surname>Widom</surname> <given-names>J</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>A map of nucleosome positions in yeast at base-pair resolution</article-title><source>Nature</source><volume>486</volume><fpage>496</fpage><lpage>501</lpage><pub-id pub-id-type="doi">10.1038/nature11142</pub-id><pub-id pub-id-type="pmid">22722846</pub-id></element-citation></ref><ref id="bib16"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Cambray</surname> <given-names>G</given-names></name><name><surname>Guimaraes</surname> <given-names>JC</given-names></name><name><surname>Arkin</surname> <given-names>AP</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Evaluation of 244,000 synthetic sequences reveals design principles to optimize translation in <italic>Escherichia coli</italic></article-title><source>Nature Biotechnology</source><volume>36</volume><fpage>1005</fpage><lpage>1015</lpage><pub-id pub-id-type="doi">10.1038/nbt.4238</pub-id><pub-id pub-id-type="pmid">30247489</pub-id></element-citation></ref><ref id="bib17"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Chang</surname> <given-names>J</given-names></name><name><surname>Zhou</surname> <given-names>Y</given-names></name><name><surname>Hu</surname> <given-names>X</given-names></name><name><surname>Lam</surname> <given-names>L</given-names></name><name><surname>Henry</surname> <given-names>C</given-names></name><name><surname>Green</surname> <given-names>EM</given-names></name><name><surname>Kita</surname> <given-names>R</given-names></name><name><surname>Kobor</surname> <given-names>MS</given-names></name><name><surname>Fraser</surname> <given-names>HB</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>The molecular mechanism of a cis-regulatory adaptation in yeast</article-title><source>PLOS Genetics</source><volume>9</volume><elocation-id>e1003813</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pgen.1003813</pub-id><pub-id pub-id-type="pmid">24068973</pub-id></element-citation></ref><ref id="bib18"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Cherry</surname> <given-names>JM</given-names></name><name><surname>Hong</surname> <given-names>EL</given-names></name><name><surname>Amundsen</surname> <given-names>C</given-names></name><name><surname>Balakrishnan</surname> <given-names>R</given-names></name><name><surname>Binkley</surname> <given-names>G</given-names></name><name><surname>Chan</surname> <given-names>ET</given-names></name><name><surname>Christie</surname> <given-names>KR</given-names></name><name><surname>Costanzo</surname> <given-names>MC</given-names></name><name><surname>Dwight</surname> <given-names>SS</given-names></name><name><surname>Engel</surname> <given-names>SR</given-names></name><name><surname>Fisk</surname> <given-names>DG</given-names></name><name><surname>Hirschman</surname> <given-names>JE</given-names></name><name><surname>Hitz</surname> <given-names>BC</given-names></name><name><surname>Karra</surname> <given-names>K</given-names></name><name><surname>Krieger</surname> <given-names>CJ</given-names></name><name><surname>Miyasato</surname> <given-names>SR</given-names></name><name><surname>Nash</surname> <given-names>RS</given-names></name><name><surname>Park</surname> <given-names>J</given-names></name><name><surname>Skrzypek</surname> <given-names>MS</given-names></name><name><surname>Simison</surname> <given-names>M</given-names></name><name><surname>Weng</surname> <given-names>S</given-names></name><name><surname>Wong</surname> <given-names>ED</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>Saccharomyces genome database: the genomics resource of budding yeast</article-title><source>Nucleic Acids Research</source><volume>40</volume><fpage>D700</fpage><lpage>D705</lpage><pub-id pub-id-type="doi">10.1093/nar/gkr1029</pub-id><pub-id pub-id-type="pmid">22110037</pub-id></element-citation></ref><ref id="bib19"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Cheung</surname> <given-names>R</given-names></name><name><surname>Insigne</surname> <given-names>KD</given-names></name><name><surname>Yao</surname> <given-names>D</given-names></name><name><surname>Burghard</surname> <given-names>CP</given-names></name><name><surname>Wang</surname> <given-names>J</given-names></name><name><surname>Hsiao</surname> <given-names>YE</given-names></name><name><surname>Jones</surname> <given-names>EM</given-names></name><name><surname>Goodman</surname> <given-names>DB</given-names></name><name><surname>Xiao</surname> <given-names>X</given-names></name><name><surname>Kosuri</surname> <given-names>S</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>A multiplexed assay for exon recognition reveals that an unappreciated fraction of rare genetic variants cause Large-Effect splicing disruptions</article-title><source>Molecular Cell</source><volume>73</volume><fpage>183</fpage><lpage>194</lpage><pub-id pub-id-type="doi">10.1016/j.molcel.2018.10.037</pub-id><pub-id pub-id-type="pmid">30503770</pub-id></element-citation></ref><ref id="bib20"><element-citation publication-type="preprint"><person-group person-group-type="author"><name><surname>Choi</surname> <given-names>J</given-names></name><name><surname>Zhang</surname> <given-names>T</given-names></name><name><surname>Vu</surname> <given-names>A</given-names></name><name><surname>Ablain</surname> <given-names>J</given-names></name><name><surname>Makowski</surname> <given-names>MM</given-names></name><name><surname>Colli</surname> <given-names>LM</given-names></name><name><surname>Xu</surname> <given-names>M</given-names></name><name><surname>Rothschild</surname> <given-names>H</given-names></name><name><surname>Gräwe</surname> <given-names>C</given-names></name><name><surname>Kovacs</surname> <given-names>MA</given-names></name><name><surname>Brossard</surname> <given-names>M</given-names></name><name><surname>Taylor</surname> <given-names>J</given-names></name><name><surname>Pasaniuc</surname> <given-names>B</given-names></name><name><surname>Chari</surname> <given-names>R</given-names></name><name><surname>Chanock</surname> <given-names>SJ</given-names></name><name><surname>Hoggart</surname> <given-names>CJ</given-names></name><name><surname>Demenais</surname> <given-names>F</given-names></name><name><surname>Barrett</surname> <given-names>JH</given-names></name><name><surname>Law</surname> <given-names>MH</given-names></name><name><surname>Iles</surname> <given-names>MM</given-names></name><name><surname>Yu</surname> <given-names>K</given-names></name><name><surname>Vermeulen</surname> <given-names>M</given-names></name><name><surname>Zon</surname> <given-names>LI</given-names></name><name><surname>Brown</surname> <given-names>KM</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Massively parallel reporter assays combined with cell-type specific eQTL informed multiple melanoma loci and identified a pleiotropic function of HIV-1 restriction gene, MX2 in melanoma promotion</article-title><source>bioRxiv</source><pub-id pub-id-type="doi">10.1101/625400</pub-id></element-citation></ref><ref id="bib21"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Claussnitzer</surname> <given-names>M</given-names></name><name><surname>Dankel</surname> <given-names>SN</given-names></name><name><surname>Kim</surname> <given-names>KH</given-names></name><name><surname>Quon</surname> <given-names>G</given-names></name><name><surname>Meuleman</surname> <given-names>W</given-names></name><name><surname>Haugen</surname> <given-names>C</given-names></name><name><surname>Glunk</surname> <given-names>V</given-names></name><name><surname>Sousa</surname> <given-names>IS</given-names></name><name><surname>Beaudry</surname> <given-names>JL</given-names></name><name><surname>Puviindran</surname> <given-names>V</given-names></name><name><surname>Abdennur</surname> <given-names>NA</given-names></name><name><surname>Liu</surname> <given-names>J</given-names></name><name><surname>Svensson</surname> <given-names>PA</given-names></name><name><surname>Hsu</surname> <given-names>YH</given-names></name><name><surname>Drucker</surname> <given-names>DJ</given-names></name><name><surname>Mellgren</surname> <given-names>G</given-names></name><name><surname>Hui</surname> <given-names>CC</given-names></name><name><surname>Hauner</surname> <given-names>H</given-names></name><name><surname>Kellis</surname> <given-names>M</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title><italic>FTO</italic> obesity variant circuitry and Adipocyte Browning in humans</article-title><source>New England Journal of Medicine</source><volume>373</volume><fpage>895</fpage><lpage>907</lpage><pub-id pub-id-type="doi">10.1056/NEJMoa1502214</pub-id><pub-id pub-id-type="pmid">26287746</pub-id></element-citation></ref><ref id="bib22"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Costanzo</surname> <given-names>M</given-names></name><name><surname>VanderSluis</surname> <given-names>B</given-names></name><name><surname>Koch</surname> <given-names>EN</given-names></name><name><surname>Baryshnikova</surname> <given-names>A</given-names></name><name><surname>Pons</surname> <given-names>C</given-names></name><name><surname>Tan</surname> <given-names>G</given-names></name><name><surname>Wang</surname> <given-names>W</given-names></name><name><surname>Usaj</surname> <given-names>M</given-names></name><name><surname>Hanchard</surname> <given-names>J</given-names></name><name><surname>Lee</surname> <given-names>SD</given-names></name><name><surname>Pelechano</surname> <given-names>V</given-names></name><name><surname>Styles</surname> <given-names>EB</given-names></name><name><surname>Billmann</surname> <given-names>M</given-names></name><name><surname>van Leeuwen</surname> <given-names>J</given-names></name><name><surname>van Dyk</surname> <given-names>N</given-names></name><name><surname>Lin</surname> <given-names>ZY</given-names></name><name><surname>Kuzmin</surname> <given-names>E</given-names></name><name><surname>Nelson</surname> <given-names>J</given-names></name><name><surname>Piotrowski</surname> <given-names>JS</given-names></name><name><surname>Srikumar</surname> <given-names>T</given-names></name><name><surname>Bahr</surname> <given-names>S</given-names></name><name><surname>Chen</surname> <given-names>Y</given-names></name><name><surname>Deshpande</surname> <given-names>R</given-names></name><name><surname>Kurat</surname> <given-names>CF</given-names></name><name><surname>Li</surname> <given-names>SC</given-names></name><name><surname>Li</surname> <given-names>Z</given-names></name><name><surname>Usaj</surname> <given-names>MM</given-names></name><name><surname>Okada</surname> <given-names>H</given-names></name><name><surname>Pascoe</surname> <given-names>N</given-names></name><name><surname>San Luis</surname> <given-names>BJ</given-names></name><name><surname>Sharifpoor</surname> <given-names>S</given-names></name><name><surname>Shuteriqi</surname> <given-names>E</given-names></name><name><surname>Simpkins</surname> <given-names>SW</given-names></name><name><surname>Snider</surname> <given-names>J</given-names></name><name><surname>Suresh</surname> <given-names>HG</given-names></name><name><surname>Tan</surname> <given-names>Y</given-names></name><name><surname>Zhu</surname> <given-names>H</given-names></name><name><surname>Malod-Dognin</surname> <given-names>N</given-names></name><name><surname>Janjic</surname> <given-names>V</given-names></name><name><surname>Przulj</surname> <given-names>N</given-names></name><name><surname>Troyanskaya</surname> <given-names>OG</given-names></name><name><surname>Stagljar</surname> <given-names>I</given-names></name><name><surname>Xia</surname> <given-names>T</given-names></name><name><surname>Ohya</surname> <given-names>Y</given-names></name><name><surname>Gingras</surname> <given-names>AC</given-names></name><name><surname>Raught</surname> <given-names>B</given-names></name><name><surname>Boutros</surname> <given-names>M</given-names></name><name><surname>Steinmetz</surname> <given-names>LM</given-names></name><name><surname>Moore</surname> <given-names>CL</given-names></name><name><surname>Rosebrock</surname> <given-names>AP</given-names></name><name><surname>Caudy</surname> <given-names>AA</given-names></name><name><surname>Myers</surname> <given-names>CL</given-names></name><name><surname>Andrews</surname> <given-names>B</given-names></name><name><surname>Boone</surname> <given-names>C</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>A global genetic interaction network maps a wiring diagram of cellular function</article-title><source>Science</source><volume>353</volume><elocation-id>aaf1420</elocation-id><pub-id pub-id-type="doi">10.1126/science.aaf1420</pub-id><pub-id pub-id-type="pmid">27708008</pub-id></element-citation></ref><ref id="bib23"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Cuperus</surname> <given-names>JT</given-names></name><name><surname>Groves</surname> <given-names>B</given-names></name><name><surname>Kuchina</surname> <given-names>A</given-names></name><name><surname>Rosenberg</surname> <given-names>AB</given-names></name><name><surname>Jojic</surname> <given-names>N</given-names></name><name><surname>Fields</surname> <given-names>S</given-names></name><name><surname>Seelig</surname> <given-names>G</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>Deep learning of the regulatory grammar of yeast 5' untranslated regions from 500,000 random sequences</article-title><source>Genome Research</source><volume>27</volume><fpage>2015</fpage><lpage>2024</lpage><pub-id pub-id-type="doi">10.1101/gr.224964.117</pub-id><pub-id pub-id-type="pmid">29097404</pub-id></element-citation></ref><ref id="bib24"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Davis</surname> <given-names>JE</given-names></name><name><surname>Insigne</surname> <given-names>KD</given-names></name><name><surname>Jones</surname> <given-names>EM</given-names></name><name><surname>Hastings</surname> <given-names>QA</given-names></name><name><surname>Boldridge</surname> <given-names>WC</given-names></name><name><surname>Kosuri</surname> <given-names>S</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>Dissection of c-AMP response element architecture by using genomic and episomal massively parallel reporter assays</article-title><source>Cell Systems</source><volume>11</volume><fpage>75</fpage><lpage>85</lpage><pub-id pub-id-type="doi">10.1016/j.cels.2020.05.011</pub-id><pub-id pub-id-type="pmid">32603702</pub-id></element-citation></ref><ref id="bib25"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>de Boer</surname> <given-names>CG</given-names></name><name><surname>Vaishnav</surname> <given-names>ED</given-names></name><name><surname>Sadeh</surname> <given-names>R</given-names></name><name><surname>Abeyta</surname> <given-names>EL</given-names></name><name><surname>Friedman</surname> <given-names>N</given-names></name><name><surname>Regev</surname> <given-names>A</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>Deciphering eukaryotic gene-regulatory logic with 100 million random promoters</article-title><source>Nature Biotechnology</source><volume>38</volume><fpage>56</fpage><lpage>65</lpage><pub-id pub-id-type="doi">10.1038/s41587-019-0315-8</pub-id><pub-id pub-id-type="pmid">31792407</pub-id></element-citation></ref><ref id="bib26"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Degner</surname> <given-names>JF</given-names></name><name><surname>Pai</surname> <given-names>AA</given-names></name><name><surname>Pique-Regi</surname> <given-names>R</given-names></name><name><surname>Veyrieras</surname> <given-names>JB</given-names></name><name><surname>Gaffney</surname> <given-names>DJ</given-names></name><name><surname>Pickrell</surname> <given-names>JK</given-names></name><name><surname>De Leon</surname> <given-names>S</given-names></name><name><surname>Michelini</surname> <given-names>K</given-names></name><name><surname>Lewellen</surname> <given-names>N</given-names></name><name><surname>Crawford</surname> <given-names>GE</given-names></name><name><surname>Stephens</surname> <given-names>M</given-names></name><name><surname>Gilad</surname> <given-names>Y</given-names></name><name><surname>Pritchard</surname> <given-names>JK</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>DNase I sensitivity QTLs are a major determinant of human expression variation</article-title><source>Nature</source><volume>482</volume><fpage>390</fpage><lpage>394</lpage><pub-id pub-id-type="doi">10.1038/nature10808</pub-id><pub-id pub-id-type="pmid">22307276</pub-id></element-citation></ref><ref id="bib27"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Dvir</surname> <given-names>S</given-names></name><name><surname>Velten</surname> <given-names>L</given-names></name><name><surname>Sharon</surname> <given-names>E</given-names></name><name><surname>Zeevi</surname> <given-names>D</given-names></name><name><surname>Carey</surname> <given-names>LB</given-names></name><name><surname>Weinberger</surname> <given-names>A</given-names></name><name><surname>Segal</surname> <given-names>E</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>Deciphering the rules by which 5'-UTR sequences affect protein expression in yeast</article-title><source>PNAS</source><volume>110</volume><fpage>E2792</fpage><lpage>E2801</lpage><pub-id pub-id-type="doi">10.1073/pnas.1222534110</pub-id><pub-id pub-id-type="pmid">23832786</pub-id></element-citation></ref><ref id="bib28"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Emerson</surname> <given-names>JJ</given-names></name><name><surname>Hsieh</surname> <given-names>LC</given-names></name><name><surname>Sung</surname> <given-names>HM</given-names></name><name><surname>Wang</surname> <given-names>TY</given-names></name><name><surname>Huang</surname> <given-names>CJ</given-names></name><name><surname>Lu</surname> <given-names>HH</given-names></name><name><surname>Lu</surname> <given-names>MY</given-names></name><name><surname>Wu</surname> <given-names>SH</given-names></name><name><surname>Li</surname> <given-names>WH</given-names></name></person-group><year iso-8601-date="2010">2010</year><article-title>Natural selection on Cis and trans regulation in yeasts</article-title><source>Genome Research</source><volume>20</volume><fpage>826</fpage><lpage>836</lpage><pub-id pub-id-type="doi">10.1101/gr.101576.109</pub-id><pub-id pub-id-type="pmid">20445163</pub-id></element-citation></ref><ref id="bib29"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Eng</surname> <given-names>KH</given-names></name><name><surname>Kvitek</surname> <given-names>DJ</given-names></name><name><surname>Keles</surname> <given-names>S</given-names></name><name><surname>Gasch</surname> <given-names>AP</given-names></name></person-group><year iso-8601-date="2010">2010</year><article-title>Transient genotype-by-environment interactions following environmental shock provide a source of expression variation for essential genes</article-title><source>Genetics</source><volume>184</volume><fpage>587</fpage><lpage>593</lpage><pub-id pub-id-type="doi">10.1534/genetics.109.107268</pub-id><pub-id pub-id-type="pmid">19966067</pub-id></element-citation></ref><ref id="bib30"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Engel</surname> <given-names>SR</given-names></name><name><surname>Dietrich</surname> <given-names>FS</given-names></name><name><surname>Fisk</surname> <given-names>DG</given-names></name><name><surname>Binkley</surname> <given-names>G</given-names></name><name><surname>Balakrishnan</surname> <given-names>R</given-names></name><name><surname>Costanzo</surname> <given-names>MC</given-names></name><name><surname>Dwight</surname> <given-names>SS</given-names></name><name><surname>Hitz</surname> <given-names>BC</given-names></name><name><surname>Karra</surname> <given-names>K</given-names></name><name><surname>Nash</surname> <given-names>RS</given-names></name><name><surname>Weng</surname> <given-names>S</given-names></name><name><surname>Wong</surname> <given-names>ED</given-names></name><name><surname>Lloyd</surname> <given-names>P</given-names></name><name><surname>Skrzypek</surname> <given-names>MS</given-names></name><name><surname>Miyasato</surname> <given-names>SR</given-names></name><name><surname>Simison</surname> <given-names>M</given-names></name><name><surname>Cherry</surname> <given-names>JM</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>The reference genome sequence of <italic>Saccharomyces cerevisiae</italic> then and now</article-title><source>G3: Genes, Genomes, Genetics</source><volume>4</volume><fpage>389</fpage><lpage>398</lpage><pub-id pub-id-type="doi">10.1534/g3.113.008995</pub-id></element-citation></ref><ref id="bib31"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Farh</surname> <given-names>KK</given-names></name><name><surname>Marson</surname> <given-names>A</given-names></name><name><surname>Zhu</surname> <given-names>J</given-names></name><name><surname>Kleinewietfeld</surname> <given-names>M</given-names></name><name><surname>Housley</surname> <given-names>WJ</given-names></name><name><surname>Beik</surname> <given-names>S</given-names></name><name><surname>Shoresh</surname> <given-names>N</given-names></name><name><surname>Whitton</surname> <given-names>H</given-names></name><name><surname>Ryan</surname> <given-names>RJ</given-names></name><name><surname>Shishkin</surname> <given-names>AA</given-names></name><name><surname>Hatan</surname> <given-names>M</given-names></name><name><surname>Carrasco-Alfonso</surname> <given-names>MJ</given-names></name><name><surname>Mayer</surname> <given-names>D</given-names></name><name><surname>Luckey</surname> <given-names>CJ</given-names></name><name><surname>Patsopoulos</surname> <given-names>NA</given-names></name><name><surname>De Jager</surname> <given-names>PL</given-names></name><name><surname>Kuchroo</surname> <given-names>VK</given-names></name><name><surname>Epstein</surname> <given-names>CB</given-names></name><name><surname>Daly</surname> <given-names>MJ</given-names></name><name><surname>Hafler</surname> <given-names>DA</given-names></name><name><surname>Bernstein</surname> <given-names>BE</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>Genetic and epigenetic fine mapping of causal autoimmune disease variants</article-title><source>Nature</source><volume>518</volume><fpage>337</fpage><lpage>343</lpage><pub-id pub-id-type="doi">10.1038/nature13835</pub-id><pub-id pub-id-type="pmid">25363779</pub-id></element-citation></ref><ref id="bib32"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Forsberg</surname> <given-names>SK</given-names></name><name><surname>Bloom</surname> <given-names>JS</given-names></name><name><surname>Sadhu</surname> <given-names>MJ</given-names></name><name><surname>Kruglyak</surname> <given-names>L</given-names></name><name><surname>Carlborg</surname> <given-names>Ö</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>Accounting for genetic interactions improves modeling of individual quantitative trait phenotypes in yeast</article-title><source>Nature Genetics</source><volume>49</volume><fpage>497</fpage><lpage>503</lpage><pub-id pub-id-type="doi">10.1038/ng.3800</pub-id><pub-id pub-id-type="pmid">28250458</pub-id></element-citation></ref><ref id="bib33"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Gietz</surname> <given-names>RD</given-names></name><name><surname>Schiestl</surname> <given-names>RH</given-names></name></person-group><year iso-8601-date="2007">2007</year><article-title>High-efficiency yeast transformation using the LiAc/SS carrier DNA/PEG method</article-title><source>Nature Protocols</source><volume>2</volume><fpage>31</fpage><lpage>34</lpage><pub-id pub-id-type="doi">10.1038/nprot.2007.13</pub-id><pub-id pub-id-type="pmid">17401334</pub-id></element-citation></ref><ref id="bib34"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Gisselbrecht</surname> <given-names>SS</given-names></name><name><surname>Barrera</surname> <given-names>LA</given-names></name><name><surname>Porsch</surname> <given-names>M</given-names></name><name><surname>Aboukhalil</surname> <given-names>A</given-names></name><name><surname>Estep</surname> <given-names>PW</given-names></name><name><surname>Vedenko</surname> <given-names>A</given-names></name><name><surname>Palagi</surname> <given-names>A</given-names></name><name><surname>Kim</surname> <given-names>Y</given-names></name><name><surname>Zhu</surname> <given-names>X</given-names></name><name><surname>Busser</surname> <given-names>BW</given-names></name><name><surname>Gamble</surname> <given-names>CE</given-names></name><name><surname>Iagovitina</surname> <given-names>A</given-names></name><name><surname>Singhania</surname> <given-names>A</given-names></name><name><surname>Michelson</surname> <given-names>AM</given-names></name><name><surname>Bulyk</surname> <given-names>ML</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>Highly parallel assays of tissue-specific enhancers in whole <italic>Drosophila</italic> embryos</article-title><source>Nature Methods</source><volume>10</volume><fpage>774</fpage><lpage>780</lpage><pub-id pub-id-type="doi">10.1038/nmeth.2558</pub-id><pub-id pub-id-type="pmid">23852450</pub-id></element-citation></ref><ref id="bib35"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Goodman</surname> <given-names>DB</given-names></name><name><surname>Church</surname> <given-names>GM</given-names></name><name><surname>Kosuri</surname> <given-names>S</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>Causes and effects of N-terminal Codon bias in bacterial genes</article-title><source>Science</source><volume>342</volume><fpage>475</fpage><lpage>479</lpage><pub-id pub-id-type="doi">10.1126/science.1241934</pub-id><pub-id pub-id-type="pmid">24072823</pub-id></element-citation></ref><ref id="bib36"><element-citation publication-type="journal"><person-group person-group-type="author"><collab>GTEx Consortium</collab><collab>Laboratory, Data Analysis &amp;Coordinating Center (LDACC)—Analysis Working Group</collab><collab>Statistical Methods groups—Analysis Working Group</collab><collab>Enhancing GTEx (eGTEx) groups</collab><collab>NIH Common Fund</collab><collab>NIH/NCI</collab><collab>NIH/NHGRI</collab><collab>NIH/NIMH</collab><collab>NIH/NIDA</collab><collab>Biospecimen Collection Source Site—NDRI</collab><collab>Biospecimen Collection Source Site—RPCI</collab><collab>Biospecimen Core Resource—VARI</collab><collab>Brain Bank Repository—University of Miami Brain Endowment Bank</collab><collab>Leidos Biomedical—Project Management</collab><collab>ELSI Study</collab><collab>Genome Browser Data Integration &amp;Visualization—EBI</collab><collab>Genome Browser Data Integration &amp;Visualization—UCSC Genomics Institute, University of California Santa Cruz</collab><collab>Lead analysts:</collab><collab>Laboratory, Data Analysis &amp;Coordinating Center (LDACC):</collab><collab>NIH program management:</collab><collab>Biospecimen collection:</collab><collab>Pathology:</collab><collab>eQTL manuscript working group:</collab><name><surname>Battle</surname> <given-names>A</given-names></name><name><surname>Brown</surname> <given-names>CD</given-names></name><name><surname>Engelhardt</surname> <given-names>BE</given-names></name><name><surname>Montgomery</surname> <given-names>SB</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>Genetic effects on gene expression across human tissues</article-title><source>Nature</source><volume>550</volume><fpage>204</fpage><lpage>213</lpage><pub-id pub-id-type="doi">10.1038/nature24277</pub-id><pub-id pub-id-type="pmid">29022597</pub-id></element-citation></ref><ref id="bib37"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Haeussler</surname> <given-names>M</given-names></name><name><surname>Zweig</surname> <given-names>AS</given-names></name><name><surname>Tyner</surname> <given-names>C</given-names></name><name><surname>Speir</surname> <given-names>ML</given-names></name><name><surname>Rosenbloom</surname> <given-names>KR</given-names></name><name><surname>Raney</surname> <given-names>BJ</given-names></name><name><surname>Lee</surname> <given-names>CM</given-names></name><name><surname>Lee</surname> <given-names>BT</given-names></name><name><surname>Hinrichs</surname> <given-names>AS</given-names></name><name><surname>Gonzalez</surname> <given-names>JN</given-names></name><name><surname>Gibson</surname> <given-names>D</given-names></name><name><surname>Diekhans</surname> <given-names>M</given-names></name><name><surname>Clawson</surname> <given-names>H</given-names></name><name><surname>Casper</surname> <given-names>J</given-names></name><name><surname>Barber</surname> <given-names>GP</given-names></name><name><surname>Haussler</surname> <given-names>D</given-names></name><name><surname>Kuhn</surname> <given-names>RM</given-names></name><name><surname>Kent</surname> <given-names>WJ</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>The UCSC genome browser database: 2019 update</article-title><source>Nucleic Acids Research</source><volume>47</volume><fpage>D853</fpage><lpage>D858</lpage><pub-id pub-id-type="doi">10.1093/nar/gky1095</pub-id><pub-id pub-id-type="pmid">30407534</pub-id></element-citation></ref><ref id="bib38"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Han</surname> <given-names>M</given-names></name><name><surname>Grunstein</surname> <given-names>M</given-names></name></person-group><year iso-8601-date="1988">1988</year><article-title>Nucleosome loss activates yeast downstream promoters in vivo</article-title><source>Cell</source><volume>55</volume><fpage>1137</fpage><lpage>1145</lpage><pub-id pub-id-type="doi">10.1016/0092-8674(88)90258-9</pub-id><pub-id pub-id-type="pmid">2849508</pub-id></element-citation></ref><ref id="bib39"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hasin-Brumshtein</surname> <given-names>Y</given-names></name><name><surname>Khan</surname> <given-names>AH</given-names></name><name><surname>Hormozdiari</surname> <given-names>F</given-names></name><name><surname>Pan</surname> <given-names>C</given-names></name><name><surname>Parks</surname> <given-names>BW</given-names></name><name><surname>Petyuk</surname> <given-names>VA</given-names></name><name><surname>Piehowski</surname> <given-names>PD</given-names></name><name><surname>Brümmer</surname> <given-names>A</given-names></name><name><surname>Pellegrini</surname> <given-names>M</given-names></name><name><surname>Xiao</surname> <given-names>X</given-names></name><name><surname>Eskin</surname> <given-names>E</given-names></name><name><surname>Smith</surname> <given-names>RD</given-names></name><name><surname>Lusis</surname> <given-names>AJ</given-names></name><name><surname>Smith</surname> <given-names>DJ</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Hypothalamic transcriptomes of 99 mouse strains reveal trans eQTL hotspots, splicing QTLs and novel non-coding genes</article-title><source>eLife</source><volume>5</volume><elocation-id>e15614</elocation-id><pub-id pub-id-type="doi">10.7554/eLife.15614</pub-id></element-citation></ref><ref id="bib40"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Heyne</surname> <given-names>HO</given-names></name><name><surname>Lautenschläger</surname> <given-names>S</given-names></name><name><surname>Nelson</surname> <given-names>R</given-names></name><name><surname>Besnier</surname> <given-names>F</given-names></name><name><surname>Rotival</surname> <given-names>M</given-names></name><name><surname>Cagan</surname> <given-names>A</given-names></name><name><surname>Kozhemyakina</surname> <given-names>R</given-names></name><name><surname>Plyusnina</surname> <given-names>IZ</given-names></name><name><surname>Trut</surname> <given-names>L</given-names></name><name><surname>Carlborg</surname> <given-names>Ö</given-names></name><name><surname>Petretto</surname> <given-names>E</given-names></name><name><surname>Kruglyak</surname> <given-names>L</given-names></name><name><surname>Pääbo</surname> <given-names>S</given-names></name><name><surname>Schöneberg</surname> <given-names>T</given-names></name><name><surname>Albert</surname> <given-names>FW</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>Genetic influences on brain gene expression in rats selected for tameness and aggression</article-title><source>Genetics</source><volume>198</volume><fpage>1277</fpage><lpage>1290</lpage><pub-id pub-id-type="doi">10.1534/genetics.114.168948</pub-id><pub-id pub-id-type="pmid">25189874</pub-id></element-citation></ref><ref id="bib41"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hill</surname> <given-names>WG</given-names></name><name><surname>Goddard</surname> <given-names>ME</given-names></name><name><surname>Visscher</surname> <given-names>PM</given-names></name></person-group><year iso-8601-date="2008">2008</year><article-title>Data and theory point to mainly additive genetic variance for complex traits</article-title><source>PLOS Genetics</source><volume>4</volume><elocation-id>e1000008</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pgen.1000008</pub-id><pub-id pub-id-type="pmid">18454194</pub-id></element-citation></ref><ref id="bib42"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Huang</surname> <given-names>YF</given-names></name><name><surname>Gulko</surname> <given-names>B</given-names></name><name><surname>Siepel</surname> <given-names>A</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>Fast, scalable prediction of deleterious noncoding variants from functional and population genomic data</article-title><source>Nature Genetics</source><volume>49</volume><fpage>618</fpage><lpage>624</lpage><pub-id pub-id-type="doi">10.1038/ng.3810</pub-id><pub-id pub-id-type="pmid">28288115</pub-id></element-citation></ref><ref id="bib43"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Huber</surname> <given-names>W</given-names></name><name><surname>Carey</surname> <given-names>VJ</given-names></name><name><surname>Gentleman</surname> <given-names>R</given-names></name><name><surname>Anders</surname> <given-names>S</given-names></name><name><surname>Carlson</surname> <given-names>M</given-names></name><name><surname>Carvalho</surname> <given-names>BS</given-names></name><name><surname>Bravo</surname> <given-names>HC</given-names></name><name><surname>Davis</surname> <given-names>S</given-names></name><name><surname>Gatto</surname> <given-names>L</given-names></name><name><surname>Girke</surname> <given-names>T</given-names></name><name><surname>Gottardo</surname> <given-names>R</given-names></name><name><surname>Hahne</surname> <given-names>F</given-names></name><name><surname>Hansen</surname> <given-names>KD</given-names></name><name><surname>Irizarry</surname> <given-names>RA</given-names></name><name><surname>Lawrence</surname> <given-names>M</given-names></name><name><surname>Love</surname> <given-names>MI</given-names></name><name><surname>MacDonald</surname> <given-names>J</given-names></name><name><surname>Obenchain</surname> <given-names>V</given-names></name><name><surname>Oleś</surname> <given-names>AK</given-names></name><name><surname>Pagès</surname> <given-names>H</given-names></name><name><surname>Reyes</surname> <given-names>A</given-names></name><name><surname>Shannon</surname> <given-names>P</given-names></name><name><surname>Smyth</surname> <given-names>GK</given-names></name><name><surname>Tenenbaum</surname> <given-names>D</given-names></name><name><surname>Waldron</surname> <given-names>L</given-names></name><name><surname>Morgan</surname> <given-names>M</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>Orchestrating high-throughput genomic analysis with bioconductor</article-title><source>Nature Methods</source><volume>12</volume><fpage>115</fpage><lpage>121</lpage><pub-id pub-id-type="doi">10.1038/nmeth.3252</pub-id><pub-id pub-id-type="pmid">25633503</pub-id></element-citation></ref><ref id="bib44"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Inoue</surname> <given-names>F</given-names></name><name><surname>Kircher</surname> <given-names>M</given-names></name><name><surname>Martin</surname> <given-names>B</given-names></name><name><surname>Cooper</surname> <given-names>GM</given-names></name><name><surname>Witten</surname> <given-names>DM</given-names></name><name><surname>McManus</surname> <given-names>MT</given-names></name><name><surname>Ahituv</surname> <given-names>N</given-names></name><name><surname>Shendure</surname> <given-names>J</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>A systematic comparison reveals substantial differences in chromosomal versus episomal encoding of enhancer activity</article-title><source>Genome Research</source><volume>27</volume><fpage>38</fpage><lpage>52</lpage><pub-id pub-id-type="doi">10.1101/gr.212092.116</pub-id><pub-id pub-id-type="pmid">27831498</pub-id></element-citation></ref><ref id="bib45"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Inoue</surname> <given-names>F</given-names></name><name><surname>Kreimer</surname> <given-names>A</given-names></name><name><surname>Ashuach</surname> <given-names>T</given-names></name><name><surname>Ahituv</surname> <given-names>N</given-names></name><name><surname>Yosef</surname> <given-names>N</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Identification and massively parallel characterization of regulatory elements driving neural induction</article-title><source>Cell Stem Cell</source><volume>25</volume><fpage>713</fpage><lpage>727</lpage><pub-id pub-id-type="doi">10.1016/j.stem.2019.09.010</pub-id><pub-id pub-id-type="pmid">31631012</pub-id></element-citation></ref><ref id="bib46"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Josephs</surname> <given-names>EB</given-names></name><name><surname>Lee</surname> <given-names>YW</given-names></name><name><surname>Stinchcombe</surname> <given-names>JR</given-names></name><name><surname>Wright</surname> <given-names>SI</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>Association mapping reveals the role of purifying selection in the maintenance of genomic variation in gene expression</article-title><source>PNAS</source><volume>112</volume><fpage>15390</fpage><lpage>15395</lpage><pub-id pub-id-type="doi">10.1073/pnas.1503027112</pub-id><pub-id pub-id-type="pmid">26604315</pub-id></element-citation></ref><ref id="bib47"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kasowski</surname> <given-names>M</given-names></name><name><surname>Kyriazopoulou-Panagiotopoulou</surname> <given-names>S</given-names></name><name><surname>Grubert</surname> <given-names>F</given-names></name><name><surname>Zaugg</surname> <given-names>JB</given-names></name><name><surname>Kundaje</surname> <given-names>A</given-names></name><name><surname>Liu</surname> <given-names>Y</given-names></name><name><surname>Boyle</surname> <given-names>AP</given-names></name><name><surname>Zhang</surname> <given-names>QC</given-names></name><name><surname>Zakharia</surname> <given-names>F</given-names></name><name><surname>Spacek</surname> <given-names>DV</given-names></name><name><surname>Li</surname> <given-names>J</given-names></name><name><surname>Xie</surname> <given-names>D</given-names></name><name><surname>Olarerin-George</surname> <given-names>A</given-names></name><name><surname>Steinmetz</surname> <given-names>LM</given-names></name><name><surname>Hogenesch</surname> <given-names>JB</given-names></name><name><surname>Kellis</surname> <given-names>M</given-names></name><name><surname>Batzoglou</surname> <given-names>S</given-names></name><name><surname>Snyder</surname> <given-names>M</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>Extensive variation in chromatin states across humans</article-title><source>Science</source><volume>342</volume><fpage>750</fpage><lpage>752</lpage><pub-id pub-id-type="doi">10.1126/science.1242510</pub-id><pub-id pub-id-type="pmid">24136358</pub-id></element-citation></ref><ref id="bib48"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kheradpour</surname> <given-names>P</given-names></name><name><surname>Ernst</surname> <given-names>J</given-names></name><name><surname>Melnikov</surname> <given-names>A</given-names></name><name><surname>Rogov</surname> <given-names>P</given-names></name><name><surname>Wang</surname> <given-names>L</given-names></name><name><surname>Zhang</surname> <given-names>X</given-names></name><name><surname>Alston</surname> <given-names>J</given-names></name><name><surname>Mikkelsen</surname> <given-names>TS</given-names></name><name><surname>Kellis</surname> <given-names>M</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>Systematic dissection of regulatory motifs in 2000 predicted human enhancers using a massively parallel reporter assay</article-title><source>Genome Research</source><volume>23</volume><fpage>800</fpage><lpage>811</lpage><pub-id pub-id-type="doi">10.1101/gr.144899.112</pub-id><pub-id pub-id-type="pmid">23512712</pub-id></element-citation></ref><ref id="bib49"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kilpinen</surname> <given-names>H</given-names></name><name><surname>Waszak</surname> <given-names>SM</given-names></name><name><surname>Gschwind</surname> <given-names>AR</given-names></name><name><surname>Raghav</surname> <given-names>SK</given-names></name><name><surname>Witwicki</surname> <given-names>RM</given-names></name><name><surname>Orioli</surname> <given-names>A</given-names></name><name><surname>Migliavacca</surname> <given-names>E</given-names></name><name><surname>Wiederkehr</surname> <given-names>M</given-names></name><name><surname>Gutierrez-Arcelus</surname> <given-names>M</given-names></name><name><surname>Panousis</surname> <given-names>NI</given-names></name><name><surname>Yurovsky</surname> <given-names>A</given-names></name><name><surname>Lappalainen</surname> <given-names>T</given-names></name><name><surname>Romano-Palumbo</surname> <given-names>L</given-names></name><name><surname>Planchon</surname> <given-names>A</given-names></name><name><surname>Bielser</surname> <given-names>D</given-names></name><name><surname>Bryois</surname> <given-names>J</given-names></name><name><surname>Padioleau</surname> <given-names>I</given-names></name><name><surname>Udin</surname> <given-names>G</given-names></name><name><surname>Thurnheer</surname> <given-names>S</given-names></name><name><surname>Hacker</surname> <given-names>D</given-names></name><name><surname>Core</surname> <given-names>LJ</given-names></name><name><surname>Lis</surname> <given-names>JT</given-names></name><name><surname>Hernandez</surname> <given-names>N</given-names></name><name><surname>Reymond</surname> <given-names>A</given-names></name><name><surname>Deplancke</surname> <given-names>B</given-names></name><name><surname>Dermitzakis</surname> <given-names>ET</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>Coordinated effects of sequence variation on DNA binding, chromatin structure, and transcription</article-title><source>Science</source><volume>342</volume><fpage>744</fpage><lpage>747</lpage><pub-id pub-id-type="doi">10.1126/science.1242463</pub-id><pub-id pub-id-type="pmid">24136355</pub-id></element-citation></ref><ref id="bib50"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kinney</surname> <given-names>JB</given-names></name><name><surname>Murugan</surname> <given-names>A</given-names></name><name><surname>Callan</surname> <given-names>CG</given-names></name><name><surname>Cox</surname> <given-names>EC</given-names></name></person-group><year iso-8601-date="2010">2010</year><article-title>Using deep sequencing to characterize the biophysical mechanism of a transcriptional regulatory sequence</article-title><source>PNAS</source><volume>107</volume><fpage>9158</fpage><lpage>9163</lpage><pub-id pub-id-type="doi">10.1073/pnas.1004290107</pub-id><pub-id pub-id-type="pmid">20439748</pub-id></element-citation></ref><ref id="bib51"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kircher</surname> <given-names>M</given-names></name><name><surname>Witten</surname> <given-names>DM</given-names></name><name><surname>Jain</surname> <given-names>P</given-names></name><name><surname>O'Roak</surname> <given-names>BJ</given-names></name><name><surname>Cooper</surname> <given-names>GM</given-names></name><name><surname>Shendure</surname> <given-names>J</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>A general framework for estimating the relative pathogenicity of human genetic variants</article-title><source>Nature Genetics</source><volume>46</volume><fpage>310</fpage><lpage>315</lpage><pub-id pub-id-type="doi">10.1038/ng.2892</pub-id><pub-id pub-id-type="pmid">24487276</pub-id></element-citation></ref><ref id="bib52"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kircher</surname> <given-names>M</given-names></name><name><surname>Xiong</surname> <given-names>C</given-names></name><name><surname>Martin</surname> <given-names>B</given-names></name><name><surname>Schubach</surname> <given-names>M</given-names></name><name><surname>Inoue</surname> <given-names>F</given-names></name><name><surname>Bell</surname> <given-names>RJA</given-names></name><name><surname>Costello</surname> <given-names>JF</given-names></name><name><surname>Shendure</surname> <given-names>J</given-names></name><name><surname>Ahituv</surname> <given-names>N</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Saturation mutagenesis of twenty disease-associated regulatory elements at single base-pair resolution</article-title><source>Nature Communications</source><volume>10</volume><fpage>1</fpage><lpage>15</lpage><pub-id pub-id-type="doi">10.1038/s41467-019-11526-w</pub-id><pub-id pub-id-type="pmid">31395865</pub-id></element-citation></ref><ref id="bib53"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kita</surname> <given-names>R</given-names></name><name><surname>Venkataram</surname> <given-names>S</given-names></name><name><surname>Zhou</surname> <given-names>Y</given-names></name><name><surname>Fraser</surname> <given-names>HB</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>High-resolution mapping of <italic>cis</italic>-regulatory variation in budding yeast</article-title><source>PNAS</source><volume>114</volume><fpage>E10736</fpage><lpage>E10744</lpage><pub-id pub-id-type="doi">10.1073/pnas.1717421114</pub-id><pub-id pub-id-type="pmid">29183975</pub-id></element-citation></ref><ref id="bib54"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Klein</surname> <given-names>JC</given-names></name><name><surname>Keith</surname> <given-names>A</given-names></name><name><surname>Rice</surname> <given-names>SJ</given-names></name><name><surname>Shepherd</surname> <given-names>C</given-names></name><name><surname>Agarwal</surname> <given-names>V</given-names></name><name><surname>Loughlin</surname> <given-names>J</given-names></name><name><surname>Shendure</surname> <given-names>J</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Functional testing of thousands of osteoarthritis-associated variants for regulatory activity</article-title><source>Nature Communications</source><volume>10</volume><fpage>1</fpage><lpage>9</lpage><pub-id pub-id-type="doi">10.1038/s41467-019-10439-y</pub-id></element-citation></ref><ref id="bib55"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kosuri</surname> <given-names>S</given-names></name><name><surname>Goodman</surname> <given-names>DB</given-names></name><name><surname>Cambray</surname> <given-names>G</given-names></name><name><surname>Mutalik</surname> <given-names>VK</given-names></name><name><surname>Gao</surname> <given-names>Y</given-names></name><name><surname>Arkin</surname> <given-names>AP</given-names></name><name><surname>Endy</surname> <given-names>D</given-names></name><name><surname>Church</surname> <given-names>GM</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>Composability of regulatory sequences controlling transcription and translation in <italic>Escherichia coli</italic></article-title><source>PNAS</source><volume>110</volume><fpage>14024</fpage><lpage>14029</lpage><pub-id pub-id-type="doi">10.1073/pnas.1301301110</pub-id><pub-id pub-id-type="pmid">23924614</pub-id></element-citation></ref><ref id="bib56"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kotopka</surname> <given-names>BJ</given-names></name><name><surname>Smolke</surname> <given-names>CD</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>Model-driven generation of artificial yeast promoters</article-title><source>Nature Communications</source><volume>11</volume><fpage>1</fpage><lpage>13</lpage><pub-id pub-id-type="doi">10.1038/s41467-020-15977-4</pub-id><pub-id pub-id-type="pmid">32355169</pub-id></element-citation></ref><ref id="bib57"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Krebs</surname> <given-names>AR</given-names></name><name><surname>Dessus-Babus</surname> <given-names>S</given-names></name><name><surname>Burger</surname> <given-names>L</given-names></name><name><surname>Schübeler</surname> <given-names>D</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>High-throughput engineering of a mammalian genome reveals building principles of methylation states at CG rich regions</article-title><source>eLife</source><volume>3</volume><elocation-id>e04094</elocation-id><pub-id pub-id-type="doi">10.7554/eLife.04094</pub-id><pub-id pub-id-type="pmid">25259795</pub-id></element-citation></ref><ref id="bib58"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kroymann</surname> <given-names>J</given-names></name><name><surname>Mitchell-Olds</surname> <given-names>T</given-names></name></person-group><year iso-8601-date="2005">2005</year><article-title>Epistasis and balanced polymorphism influencing complex trait variation</article-title><source>Nature</source><volume>435</volume><fpage>95</fpage><lpage>98</lpage><pub-id pub-id-type="doi">10.1038/nature03480</pub-id><pub-id pub-id-type="pmid">15875023</pub-id></element-citation></ref><ref id="bib59"><element-citation publication-type="book"><person-group person-group-type="author"><name><surname>Kuhn</surname> <given-names>M</given-names></name><name><surname>Wing</surname> <given-names>J</given-names></name><name><surname>Weston</surname> <given-names>S</given-names></name><name><surname>Williams</surname> <given-names>A</given-names></name><name><surname>Keefer</surname> <given-names>C</given-names></name><name><surname>Engelhardt</surname> <given-names>A</given-names></name><name><surname>Cooper</surname> <given-names>T</given-names></name><name><surname>Mayer</surname> <given-names>Z</given-names></name><name><surname>Kenkel</surname> <given-names>B</given-names></name><name><surname>Core Team</surname> <given-names>R</given-names></name><name><surname>Benesty</surname> <given-names>M</given-names></name><name><surname>Lescarbeau</surname> <given-names>R</given-names></name><name><surname>Ziem</surname> <given-names>A</given-names></name><name><surname>Scrucca</surname> <given-names>L</given-names></name><name><surname>Tang</surname> <given-names>Y</given-names></name><name><surname>Candan</surname> <given-names>C</given-names></name><name><surname>Hunt</surname> <given-names>T</given-names></name></person-group><year iso-8601-date="2020">2020</year><source>Caret: Classification and Regression Training</source><publisher-name>Astrophysics Source Code Library</publisher-name></element-citation></ref><ref id="bib60"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kwasnieski</surname> <given-names>JC</given-names></name><name><surname>Mogno</surname> <given-names>I</given-names></name><name><surname>Myers</surname> <given-names>CA</given-names></name><name><surname>Corbo</surname> <given-names>JC</given-names></name><name><surname>Cohen</surname> <given-names>BA</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>Complex effects of nucleotide variants in a mammalian cis-regulatory element</article-title><source>PNAS</source><volume>109</volume><fpage>19498</fpage><lpage>19503</lpage><pub-id pub-id-type="doi">10.1073/pnas.1210678109</pub-id><pub-id pub-id-type="pmid">23129659</pub-id></element-citation></ref><ref id="bib61"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kwasnieski</surname> <given-names>JC</given-names></name><name><surname>Fiore</surname> <given-names>C</given-names></name><name><surname>Chaudhari</surname> <given-names>HG</given-names></name><name><surname>Cohen</surname> <given-names>BA</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>High-throughput functional testing of ENCODE segmentation predictions</article-title><source>Genome Research</source><volume>24</volume><fpage>1595</fpage><lpage>1602</lpage><pub-id pub-id-type="doi">10.1101/gr.173518.114</pub-id><pub-id pub-id-type="pmid">25035418</pub-id></element-citation></ref><ref id="bib62"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Lee</surname> <given-names>D</given-names></name><name><surname>Gorkin</surname> <given-names>DU</given-names></name><name><surname>Baker</surname> <given-names>M</given-names></name><name><surname>Strober</surname> <given-names>BJ</given-names></name><name><surname>Asoni</surname> <given-names>AL</given-names></name><name><surname>McCallion</surname> <given-names>AS</given-names></name><name><surname>Beer</surname> <given-names>MA</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>A method to predict the impact of regulatory variants from DNA sequence</article-title><source>Nature Genetics</source><volume>47</volume><fpage>955</fpage><lpage>961</lpage><pub-id pub-id-type="doi">10.1038/ng.3331</pub-id></element-citation></ref><ref id="bib63"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Lin</surname> <given-names>Z</given-names></name><name><surname>Wu</surname> <given-names>WS</given-names></name><name><surname>Liang</surname> <given-names>H</given-names></name><name><surname>Woo</surname> <given-names>Y</given-names></name><name><surname>Li</surname> <given-names>WH</given-names></name></person-group><year iso-8601-date="2010">2010</year><article-title>The spatial distribution of Cis regulatory elements in yeast promoters and its implications for transcriptional regulation</article-title><source>BMC Genomics</source><volume>11</volume><elocation-id>581</elocation-id><pub-id pub-id-type="doi">10.1186/1471-2164-11-581</pub-id><pub-id pub-id-type="pmid">20958978</pub-id></element-citation></ref><ref id="bib64"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>S</given-names></name><name><surname>Liu</surname> <given-names>Y</given-names></name><name><surname>Zhang</surname> <given-names>Q</given-names></name><name><surname>Wu</surname> <given-names>J</given-names></name><name><surname>Liang</surname> <given-names>J</given-names></name><name><surname>Yu</surname> <given-names>S</given-names></name><name><surname>Wei</surname> <given-names>GH</given-names></name><name><surname>White</surname> <given-names>KP</given-names></name><name><surname>Wang</surname> <given-names>X</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>Systematic identification of regulatory variants associated with Cancer risk</article-title><source>Genome Biology</source><volume>18</volume><elocation-id>194</elocation-id><pub-id pub-id-type="doi">10.1186/s13059-017-1322-z</pub-id><pub-id pub-id-type="pmid">29061142</pub-id></element-citation></ref><ref id="bib65"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>L</given-names></name><name><surname>Sanderford</surname> <given-names>MD</given-names></name><name><surname>Patel</surname> <given-names>R</given-names></name><name><surname>Chandrashekar</surname> <given-names>P</given-names></name><name><surname>Gibson</surname> <given-names>G</given-names></name><name><surname>Kumar</surname> <given-names>S</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Biological relevance of computationally predicted pathogenicity of noncoding variants</article-title><source>Nature Communications</source><volume>10</volume><fpage>1</fpage><lpage>11</lpage><pub-id pub-id-type="doi">10.1038/s41467-018-08270-y</pub-id></element-citation></ref><ref id="bib66"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Lubliner</surname> <given-names>S</given-names></name><name><surname>Regev</surname> <given-names>I</given-names></name><name><surname>Lotan-Pompan</surname> <given-names>M</given-names></name><name><surname>Edelheit</surname> <given-names>S</given-names></name><name><surname>Weinberger</surname> <given-names>A</given-names></name><name><surname>Segal</surname> <given-names>E</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>Core promoter sequence in yeast is a major determinant of expression level</article-title><source>Genome Research</source><volume>25</volume><fpage>1008</fpage><lpage>1017</lpage><pub-id pub-id-type="doi">10.1101/gr.188193.114</pub-id><pub-id pub-id-type="pmid">25969468</pub-id></element-citation></ref><ref id="bib67"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Lutz</surname> <given-names>S</given-names></name><name><surname>Brion</surname> <given-names>C</given-names></name><name><surname>Kliebhan</surname> <given-names>M</given-names></name><name><surname>Albert</surname> <given-names>FW</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>DNA variants affecting the expression of numerous genes in trans have diverse mechanisms of action and evolutionary histories</article-title><source>PLOS Genetics</source><volume>15</volume><elocation-id>e1008375</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pgen.1008375</pub-id><pub-id pub-id-type="pmid">31738765</pub-id></element-citation></ref><ref id="bib68"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Mackay</surname> <given-names>TF</given-names></name><name><surname>Stone</surname> <given-names>EA</given-names></name><name><surname>Ayroles</surname> <given-names>JF</given-names></name></person-group><year iso-8601-date="2009">2009</year><article-title>The genetics of quantitative traits: challenges and prospects</article-title><source>Nature Reviews Genetics</source><volume>10</volume><fpage>565</fpage><lpage>577</lpage><pub-id pub-id-type="doi">10.1038/nrg2612</pub-id><pub-id pub-id-type="pmid">19584810</pub-id></element-citation></ref><ref id="bib69"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Maricque</surname> <given-names>BB</given-names></name><name><surname>Chaudhari</surname> <given-names>HG</given-names></name><name><surname>Cohen</surname> <given-names>BA</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>A massively parallel reporter assay dissects the influence of chromatin structure on cis-regulatory activity</article-title><source>Nature Biotechnology</source><volume>37</volume><fpage>90</fpage><lpage>95</lpage><pub-id pub-id-type="doi">10.1038/nbt.4285</pub-id></element-citation></ref><ref id="bib70"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Matreyek</surname> <given-names>KA</given-names></name><name><surname>Starita</surname> <given-names>LM</given-names></name><name><surname>Stephany</surname> <given-names>JJ</given-names></name><name><surname>Martin</surname> <given-names>B</given-names></name><name><surname>Chiasson</surname> <given-names>MA</given-names></name><name><surname>Gray</surname> <given-names>VE</given-names></name><name><surname>Kircher</surname> <given-names>M</given-names></name><name><surname>Khechaduri</surname> <given-names>A</given-names></name><name><surname>Dines</surname> <given-names>JN</given-names></name><name><surname>Hause</surname> <given-names>RJ</given-names></name><name><surname>Bhatia</surname> <given-names>S</given-names></name><name><surname>Evans</surname> <given-names>WE</given-names></name><name><surname>Relling</surname> <given-names>MV</given-names></name><name><surname>Yang</surname> <given-names>W</given-names></name><name><surname>Shendure</surname> <given-names>J</given-names></name><name><surname>Fowler</surname> <given-names>DM</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Multiplex assessment of protein variant abundance by massively parallel sequencing</article-title><source>Nature Genetics</source><volume>50</volume><fpage>874</fpage><lpage>882</lpage><pub-id pub-id-type="doi">10.1038/s41588-018-0122-z</pub-id><pub-id pub-id-type="pmid">29785012</pub-id></element-citation></ref><ref id="bib71"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Maurer</surname> <given-names>MJ</given-names></name><name><surname>Sutardja</surname> <given-names>L</given-names></name><name><surname>Pinel</surname> <given-names>D</given-names></name><name><surname>Bauer</surname> <given-names>S</given-names></name><name><surname>Muehlbauer</surname> <given-names>AL</given-names></name><name><surname>Ames</surname> <given-names>TD</given-names></name><name><surname>Skerker</surname> <given-names>JM</given-names></name><name><surname>Arkin</surname> <given-names>AP</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>Quantitative trait loci (QTL)-Guided metabolic engineering of a complex trait</article-title><source>ACS Synthetic Biology</source><volume>6</volume><fpage>566</fpage><lpage>581</lpage><pub-id pub-id-type="doi">10.1021/acssynbio.6b00264</pub-id><pub-id pub-id-type="pmid">27936603</pub-id></element-citation></ref><ref id="bib72"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>McVicker</surname> <given-names>G</given-names></name><name><surname>van de Geijn</surname> <given-names>B</given-names></name><name><surname>Degner</surname> <given-names>JF</given-names></name><name><surname>Cain</surname> <given-names>CE</given-names></name><name><surname>Banovich</surname> <given-names>NE</given-names></name><name><surname>Raj</surname> <given-names>A</given-names></name><name><surname>Lewellen</surname> <given-names>N</given-names></name><name><surname>Myrthil</surname> <given-names>M</given-names></name><name><surname>Gilad</surname> <given-names>Y</given-names></name><name><surname>Pritchard</surname> <given-names>JK</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>Identification of genetic variants that affect histone modifications in human cells</article-title><source>Science</source><volume>342</volume><fpage>747</fpage><lpage>749</lpage><pub-id pub-id-type="doi">10.1126/science.1242429</pub-id><pub-id pub-id-type="pmid">24136359</pub-id></element-citation></ref><ref id="bib73"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Melnikov</surname> <given-names>A</given-names></name><name><surname>Murugan</surname> <given-names>A</given-names></name><name><surname>Zhang</surname> <given-names>X</given-names></name><name><surname>Tesileanu</surname> <given-names>T</given-names></name><name><surname>Wang</surname> <given-names>L</given-names></name><name><surname>Rogov</surname> <given-names>P</given-names></name><name><surname>Feizi</surname> <given-names>S</given-names></name><name><surname>Gnirke</surname> <given-names>A</given-names></name><name><surname>Callan</surname> <given-names>CG</given-names></name><name><surname>Kinney</surname> <given-names>JB</given-names></name><name><surname>Kellis</surname> <given-names>M</given-names></name><name><surname>Lander</surname> <given-names>ES</given-names></name><name><surname>Mikkelsen</surname> <given-names>TS</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>Systematic dissection and optimization of inducible enhancers in human cells using a massively parallel reporter assay</article-title><source>Nature Biotechnology</source><volume>30</volume><fpage>271</fpage><lpage>277</lpage><pub-id pub-id-type="doi">10.1038/nbt.2137</pub-id><pub-id pub-id-type="pmid">22371084</pub-id></element-citation></ref><ref id="bib74"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Metzger</surname> <given-names>BPH</given-names></name><name><surname>Wittkopp</surname> <given-names>PJ</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Compensatory <italic>trans</italic>-regulatory alleles minimizing variation in <italic>TDH3</italic> expression are common within <italic>Saccharomyces cerevisiae</italic></article-title><source>Evolution Letters</source><volume>3</volume><fpage>448</fpage><lpage>461</lpage><pub-id pub-id-type="doi">10.1002/evl3.137</pub-id><pub-id pub-id-type="pmid">31636938</pub-id></element-citation></ref><ref id="bib75"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Mogno</surname> <given-names>I</given-names></name><name><surname>Kwasnieski</surname> <given-names>JC</given-names></name><name><surname>Cohen</surname> <given-names>BA</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>Massively parallel synthetic promoter assays reveal the in vivo effects of binding site variants</article-title><source>Genome Research</source><volume>23</volume><fpage>1908</fpage><lpage>1915</lpage><pub-id pub-id-type="doi">10.1101/gr.157891.113</pub-id><pub-id pub-id-type="pmid">23921661</pub-id></element-citation></ref><ref id="bib76"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Monteiro</surname> <given-names>PT</given-names></name><name><surname>Oliveira</surname> <given-names>J</given-names></name><name><surname>Pais</surname> <given-names>P</given-names></name><name><surname>Antunes</surname> <given-names>M</given-names></name><name><surname>Palma</surname> <given-names>M</given-names></name><name><surname>Cavalheiro</surname> <given-names>M</given-names></name><name><surname>Galocha</surname> <given-names>M</given-names></name><name><surname>Godinho</surname> <given-names>CP</given-names></name><name><surname>Martins</surname> <given-names>LC</given-names></name><name><surname>Bourbon</surname> <given-names>N</given-names></name><name><surname>Mota</surname> <given-names>MN</given-names></name><name><surname>Ribeiro</surname> <given-names>RA</given-names></name><name><surname>Viana</surname> <given-names>R</given-names></name><name><surname>Sá-Correia</surname> <given-names>I</given-names></name><name><surname>Teixeira</surname> <given-names>MC</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>YEASTRACT+: a portal for cross-species comparative genomics of transcription regulation in yeasts</article-title><source>Nucleic Acids Research</source><volume>48</volume><fpage>D642</fpage><lpage>D649</lpage><pub-id pub-id-type="doi">10.1093/nar/gkz859</pub-id><pub-id pub-id-type="pmid">31586406</pub-id></element-citation></ref><ref id="bib77"><element-citation publication-type="preprint"><person-group person-group-type="author"><name><surname>Mulvey</surname> <given-names>B</given-names></name><name><surname>Lagunas</surname> <given-names>T</given-names></name><name><surname>Dougherty</surname> <given-names>JD</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>The Oft-Overlooked massively parallel reporter assay: where, when <italic>and Which Psychiatric Genetic Variants are Functional?</italic></article-title><source>bioRxiv</source><pub-id pub-id-type="doi">10.1101/2020.02.02.931337</pub-id></element-citation></ref><ref id="bib78"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Musunuru</surname> <given-names>K</given-names></name><name><surname>Strong</surname> <given-names>A</given-names></name><name><surname>Frank-Kamenetsky</surname> <given-names>M</given-names></name><name><surname>Lee</surname> <given-names>NE</given-names></name><name><surname>Ahfeldt</surname> <given-names>T</given-names></name><name><surname>Sachs</surname> <given-names>KV</given-names></name><name><surname>Li</surname> <given-names>X</given-names></name><name><surname>Li</surname> <given-names>H</given-names></name><name><surname>Kuperwasser</surname> <given-names>N</given-names></name><name><surname>Ruda</surname> <given-names>VM</given-names></name><name><surname>Pirruccello</surname> <given-names>JP</given-names></name><name><surname>Muchmore</surname> <given-names>B</given-names></name><name><surname>Prokunina-Olsson</surname> <given-names>L</given-names></name><name><surname>Hall</surname> <given-names>JL</given-names></name><name><surname>Schadt</surname> <given-names>EE</given-names></name><name><surname>Morales</surname> <given-names>CR</given-names></name><name><surname>Lund-Katz</surname> <given-names>S</given-names></name><name><surname>Phillips</surname> <given-names>MC</given-names></name><name><surname>Wong</surname> <given-names>J</given-names></name><name><surname>Cantley</surname> <given-names>W</given-names></name><name><surname>Racie</surname> <given-names>T</given-names></name><name><surname>Ejebe</surname> <given-names>KG</given-names></name><name><surname>Orho-Melander</surname> <given-names>M</given-names></name><name><surname>Melander</surname> <given-names>O</given-names></name><name><surname>Koteliansky</surname> <given-names>V</given-names></name><name><surname>Fitzgerald</surname> <given-names>K</given-names></name><name><surname>Krauss</surname> <given-names>RM</given-names></name><name><surname>Cowan</surname> <given-names>CA</given-names></name><name><surname>Kathiresan</surname> <given-names>S</given-names></name><name><surname>Rader</surname> <given-names>DJ</given-names></name></person-group><year iso-8601-date="2010">2010</year><article-title>From noncoding variant to phenotype via SORT1 at the 1p13 cholesterol locus</article-title><source>Nature</source><volume>466</volume><fpage>714</fpage><lpage>719</lpage><pub-id pub-id-type="doi">10.1038/nature09266</pub-id><pub-id pub-id-type="pmid">20686566</pub-id></element-citation></ref><ref id="bib79"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Myint</surname> <given-names>L</given-names></name><name><surname>Avramopoulos</surname> <given-names>DG</given-names></name><name><surname>Goff</surname> <given-names>LA</given-names></name><name><surname>Hansen</surname> <given-names>KD</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Linear models enable powerful differential activity analysis in massively parallel reporter assays</article-title><source>BMC Genomics</source><volume>20</volume><elocation-id>209</elocation-id><pub-id pub-id-type="doi">10.1186/s12864-019-5556-x</pub-id><pub-id pub-id-type="pmid">30866806</pub-id></element-citation></ref><ref id="bib80"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Newman</surname> <given-names>JR</given-names></name><name><surname>Ghaemmaghami</surname> <given-names>S</given-names></name><name><surname>Ihmels</surname> <given-names>J</given-names></name><name><surname>Breslow</surname> <given-names>DK</given-names></name><name><surname>Noble</surname> <given-names>M</given-names></name><name><surname>DeRisi</surname> <given-names>JL</given-names></name><name><surname>Weissman</surname> <given-names>JS</given-names></name></person-group><year iso-8601-date="2006">2006</year><article-title>Single-cell proteomic analysis of <italic>S. cerevisiae</italic> reveals the architecture of biological noise</article-title><source>Nature</source><volume>441</volume><fpage>840</fpage><lpage>846</lpage><pub-id pub-id-type="doi">10.1038/nature04785</pub-id><pub-id pub-id-type="pmid">16699522</pub-id></element-citation></ref><ref id="bib81"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Nishizaki</surname> <given-names>SS</given-names></name><name><surname>Boyle</surname> <given-names>AP</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>Mining the unknown: assigning function to noncoding single nucleotide polymorphisms</article-title><source>Trends in Genetics</source><volume>33</volume><fpage>34</fpage><lpage>45</lpage><pub-id pub-id-type="doi">10.1016/j.tig.2016.10.008</pub-id><pub-id pub-id-type="pmid">27939749</pub-id></element-citation></ref><ref id="bib82"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Pashos</surname> <given-names>EE</given-names></name><name><surname>Park</surname> <given-names>Y</given-names></name><name><surname>Wang</surname> <given-names>X</given-names></name><name><surname>Raghavan</surname> <given-names>A</given-names></name><name><surname>Yang</surname> <given-names>W</given-names></name><name><surname>Abbey</surname> <given-names>D</given-names></name><name><surname>Peters</surname> <given-names>DT</given-names></name><name><surname>Arbelaez</surname> <given-names>J</given-names></name><name><surname>Hernandez</surname> <given-names>M</given-names></name><name><surname>Kuperwasser</surname> <given-names>N</given-names></name><name><surname>Li</surname> <given-names>W</given-names></name><name><surname>Lian</surname> <given-names>Z</given-names></name><name><surname>Liu</surname> <given-names>Y</given-names></name><name><surname>Lv</surname> <given-names>W</given-names></name><name><surname>Lytle-Gabbin</surname> <given-names>SL</given-names></name><name><surname>Marchadier</surname> <given-names>DH</given-names></name><name><surname>Rogov</surname> <given-names>P</given-names></name><name><surname>Shi</surname> <given-names>J</given-names></name><name><surname>Slovik</surname> <given-names>KJ</given-names></name><name><surname>Stylianou</surname> <given-names>IM</given-names></name><name><surname>Wang</surname> <given-names>L</given-names></name><name><surname>Yan</surname> <given-names>R</given-names></name><name><surname>Zhang</surname> <given-names>X</given-names></name><name><surname>Kathiresan</surname> <given-names>S</given-names></name><name><surname>Duncan</surname> <given-names>SA</given-names></name><name><surname>Mikkelsen</surname> <given-names>TS</given-names></name><name><surname>Morrisey</surname> <given-names>EE</given-names></name><name><surname>Rader</surname> <given-names>DJ</given-names></name><name><surname>Brown</surname> <given-names>CD</given-names></name><name><surname>Musunuru</surname> <given-names>K</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>Large, diverse population cohorts of hiPSCs and derived Hepatocyte-like cells reveal functional genetic variation at blood Lipid-Associated loci</article-title><source>Cell Stem Cell</source><volume>20</volume><fpage>558</fpage><lpage>570</lpage><pub-id pub-id-type="doi">10.1016/j.stem.2017.03.017</pub-id><pub-id pub-id-type="pmid">28388432</pub-id></element-citation></ref><ref id="bib83"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Patwardhan</surname> <given-names>RP</given-names></name><name><surname>Lee</surname> <given-names>C</given-names></name><name><surname>Litvin</surname> <given-names>O</given-names></name><name><surname>Young</surname> <given-names>DL</given-names></name><name><surname>Pe'er</surname> <given-names>D</given-names></name><name><surname>Shendure</surname> <given-names>J</given-names></name></person-group><year iso-8601-date="2009">2009</year><article-title>High-resolution analysis of DNA regulatory elements by synthetic saturation mutagenesis</article-title><source>Nature Biotechnology</source><volume>27</volume><fpage>1173</fpage><lpage>1175</lpage><pub-id pub-id-type="doi">10.1038/nbt.1589</pub-id><pub-id pub-id-type="pmid">19915551</pub-id></element-citation></ref><ref id="bib84"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Pelechano</surname> <given-names>V</given-names></name><name><surname>Wei</surname> <given-names>W</given-names></name><name><surname>Steinmetz</surname> <given-names>LM</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>Extensive transcriptional heterogeneity revealed by isoform profiling</article-title><source>Nature</source><volume>497</volume><fpage>127</fpage><lpage>131</lpage><pub-id pub-id-type="doi">10.1038/nature12121</pub-id><pub-id pub-id-type="pmid">23615609</pub-id></element-citation></ref><ref id="bib85"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Peter</surname> <given-names>J</given-names></name><name><surname>De Chiara</surname> <given-names>M</given-names></name><name><surname>Friedrich</surname> <given-names>A</given-names></name><name><surname>Yue</surname> <given-names>JX</given-names></name><name><surname>Pflieger</surname> <given-names>D</given-names></name><name><surname>Bergström</surname> <given-names>A</given-names></name><name><surname>Sigwalt</surname> <given-names>A</given-names></name><name><surname>Barre</surname> <given-names>B</given-names></name><name><surname>Freel</surname> <given-names>K</given-names></name><name><surname>Llored</surname> <given-names>A</given-names></name><name><surname>Cruaud</surname> <given-names>C</given-names></name><name><surname>Labadie</surname> <given-names>K</given-names></name><name><surname>Aury</surname> <given-names>JM</given-names></name><name><surname>Istace</surname> <given-names>B</given-names></name><name><surname>Lebrigand</surname> <given-names>K</given-names></name><name><surname>Barbry</surname> <given-names>P</given-names></name><name><surname>Engelen</surname> <given-names>S</given-names></name><name><surname>Lemainque</surname> <given-names>A</given-names></name><name><surname>Wincker</surname> <given-names>P</given-names></name><name><surname>Liti</surname> <given-names>G</given-names></name><name><surname>Schacherer</surname> <given-names>J</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Genome evolution across 1,011 <italic>Saccharomyces cerevisiae</italic> isolates</article-title><source>Nature</source><volume>556</volume><fpage>339</fpage><lpage>344</lpage><pub-id pub-id-type="doi">10.1038/s41586-018-0030-5</pub-id><pub-id pub-id-type="pmid">29643504</pub-id></element-citation></ref><ref id="bib86"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Rabani</surname> <given-names>M</given-names></name><name><surname>Pieper</surname> <given-names>L</given-names></name><name><surname>Chew</surname> <given-names>GL</given-names></name><name><surname>Schier</surname> <given-names>AF</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>A massively parallel reporter assay of 3' UTR sequences identifies in Vivo Rules for mRNA Degradation</article-title><source>Molecular Cell</source><volume>68</volume><fpage>1083</fpage><lpage>1094</lpage><pub-id pub-id-type="doi">10.1016/j.molcel.2017.11.014</pub-id><pub-id pub-id-type="pmid">29225039</pub-id></element-citation></ref><ref id="bib87"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Rando</surname> <given-names>OJ</given-names></name><name><surname>Winston</surname> <given-names>F</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>Chromatin and transcription in yeast</article-title><source>Genetics</source><volume>190</volume><fpage>351</fpage><lpage>387</lpage><pub-id pub-id-type="doi">10.1534/genetics.111.132266</pub-id><pub-id pub-id-type="pmid">22345607</pub-id></element-citation></ref><ref id="bib88"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Rockman</surname> <given-names>MV</given-names></name><name><surname>Skrovanek</surname> <given-names>SS</given-names></name><name><surname>Kruglyak</surname> <given-names>L</given-names></name></person-group><year iso-8601-date="2010">2010</year><article-title>Selection at linked sites shapes heritable phenotypic variation in <italic>C. elegans</italic></article-title><source>Science</source><volume>330</volume><fpage>372</fpage><lpage>376</lpage><pub-id pub-id-type="doi">10.1126/science.1194208</pub-id><pub-id pub-id-type="pmid">20947766</pub-id></element-citation></ref><ref id="bib89"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ronald</surname> <given-names>J</given-names></name><name><surname>Brem</surname> <given-names>RB</given-names></name><name><surname>Whittle</surname> <given-names>J</given-names></name><name><surname>Kruglyak</surname> <given-names>L</given-names></name></person-group><year iso-8601-date="2005">2005</year><article-title>Local regulatory variation in <italic>Saccharomyces cerevisiae</italic></article-title><source>PLOS Genetics</source><volume>1</volume><elocation-id>e25</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pgen.0010025</pub-id><pub-id pub-id-type="pmid">16121257</pub-id></element-citation></ref><ref id="bib90"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ronald</surname> <given-names>J</given-names></name><name><surname>Akey</surname> <given-names>JM</given-names></name></person-group><year iso-8601-date="2007">2007</year><article-title>The evolution of gene expression QTL in <italic>Saccharomyces cerevisiae</italic></article-title><source>PLOS ONE</source><volume>2</volume><elocation-id>e678</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pone.0000678</pub-id><pub-id pub-id-type="pmid">17668057</pub-id></element-citation></ref><ref id="bib91"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Rosenberg</surname> <given-names>AB</given-names></name><name><surname>Patwardhan</surname> <given-names>RP</given-names></name><name><surname>Shendure</surname> <given-names>J</given-names></name><name><surname>Seelig</surname> <given-names>G</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>Learning the sequence determinants of alternative splicing from millions of random sequences</article-title><source>Cell</source><volume>163</volume><fpage>698</fpage><lpage>711</lpage><pub-id pub-id-type="doi">10.1016/j.cell.2015.09.054</pub-id><pub-id pub-id-type="pmid">26496609</pub-id></element-citation></ref><ref id="bib92"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Safra</surname> <given-names>M</given-names></name><name><surname>Nir</surname> <given-names>R</given-names></name><name><surname>Farouq</surname> <given-names>D</given-names></name><name><surname>Vainberg Slutskin</surname> <given-names>I</given-names></name><name><surname>Schwartz</surname> <given-names>S</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>TRUB1 is the predominant pseudouridine synthase acting on mammalian mRNA via a predictable and conserved code</article-title><source>Genome Research</source><volume>27</volume><fpage>393</fpage><lpage>406</lpage><pub-id pub-id-type="doi">10.1101/gr.207613.116</pub-id><pub-id pub-id-type="pmid">28073919</pub-id></element-citation></ref><ref id="bib93"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Shalem</surname> <given-names>O</given-names></name><name><surname>Sharon</surname> <given-names>E</given-names></name><name><surname>Lubliner</surname> <given-names>S</given-names></name><name><surname>Regev</surname> <given-names>I</given-names></name><name><surname>Lotan-Pompan</surname> <given-names>M</given-names></name><name><surname>Yakhini</surname> <given-names>Z</given-names></name><name><surname>Segal</surname> <given-names>E</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>Systematic dissection of the sequence determinants of gene 3' end mediated expression control</article-title><source>PLOS Genetics</source><volume>11</volume><elocation-id>e1005147</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pgen.1005147</pub-id><pub-id pub-id-type="pmid">25875337</pub-id></element-citation></ref><ref id="bib94"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Sharon</surname> <given-names>E</given-names></name><name><surname>Kalma</surname> <given-names>Y</given-names></name><name><surname>Sharp</surname> <given-names>A</given-names></name><name><surname>Raveh-Sadka</surname> <given-names>T</given-names></name><name><surname>Levo</surname> <given-names>M</given-names></name><name><surname>Zeevi</surname> <given-names>D</given-names></name><name><surname>Keren</surname> <given-names>L</given-names></name><name><surname>Yakhini</surname> <given-names>Z</given-names></name><name><surname>Weinberger</surname> <given-names>A</given-names></name><name><surname>Segal</surname> <given-names>E</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>Inferring gene regulatory logic from high-throughput measurements of thousands of systematically designed promoters</article-title><source>Nature Biotechnology</source><volume>30</volume><fpage>521</fpage><lpage>530</lpage><pub-id pub-id-type="doi">10.1038/nbt.2205</pub-id><pub-id pub-id-type="pmid">22609971</pub-id></element-citation></ref><ref id="bib95"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Sheff</surname> <given-names>MA</given-names></name><name><surname>Thorn</surname> <given-names>KS</given-names></name></person-group><year iso-8601-date="2004">2004</year><article-title>Optimized cassettes for fluorescent protein tagging in <italic>Saccharomyces cerevisiae</italic></article-title><source>Yeast</source><volume>21</volume><fpage>661</fpage><lpage>670</lpage><pub-id pub-id-type="doi">10.1002/yea.1130</pub-id><pub-id pub-id-type="pmid">15197731</pub-id></element-citation></ref><ref id="bib96"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Siepel</surname> <given-names>A</given-names></name><name><surname>Bejerano</surname> <given-names>G</given-names></name><name><surname>Pedersen</surname> <given-names>JS</given-names></name><name><surname>Hinrichs</surname> <given-names>AS</given-names></name><name><surname>Hou</surname> <given-names>M</given-names></name><name><surname>Rosenbloom</surname> <given-names>K</given-names></name><name><surname>Clawson</surname> <given-names>H</given-names></name><name><surname>Spieth</surname> <given-names>J</given-names></name><name><surname>Hillier</surname> <given-names>LW</given-names></name><name><surname>Richards</surname> <given-names>S</given-names></name><name><surname>Weinstock</surname> <given-names>GM</given-names></name><name><surname>Wilson</surname> <given-names>RK</given-names></name><name><surname>Gibbs</surname> <given-names>RA</given-names></name><name><surname>Kent</surname> <given-names>WJ</given-names></name><name><surname>Miller</surname> <given-names>W</given-names></name><name><surname>Haussler</surname> <given-names>D</given-names></name></person-group><year iso-8601-date="2005">2005</year><article-title>Evolutionarily conserved elements in vertebrate, insect, worm, and yeast genomes</article-title><source>Genome Research</source><volume>15</volume><fpage>1034</fpage><lpage>1050</lpage><pub-id pub-id-type="doi">10.1101/gr.3715005</pub-id><pub-id pub-id-type="pmid">16024819</pub-id></element-citation></ref><ref id="bib97"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Signor</surname> <given-names>SA</given-names></name><name><surname>Nuzhdin</surname> <given-names>SV</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>The evolution of gene expression in Cis and trans</article-title><source>Trends in Genetics</source><volume>34</volume><fpage>532</fpage><lpage>544</lpage><pub-id pub-id-type="doi">10.1016/j.tig.2018.03.007</pub-id><pub-id pub-id-type="pmid">29680748</pub-id></element-citation></ref><ref id="bib98"><element-citation publication-type="software"><person-group person-group-type="author"><name><surname>Signorell</surname> <given-names>A</given-names></name><name><surname>Aho</surname> <given-names>K</given-names></name><name><surname>Alfons</surname> <given-names>A</given-names></name><name><surname>Anderegg</surname> <given-names>N</given-names></name><name><surname>Aragon</surname> <given-names>T</given-names></name><name><surname>Arppe</surname> <given-names>A</given-names></name><name><surname>Baddeley</surname> <given-names>A</given-names></name><name><surname>Barton</surname> <given-names>K</given-names></name><name><surname>Bolker</surname> <given-names>B</given-names></name><name><surname>Borchers</surname> <given-names>HW</given-names></name><name><surname>Caeiro</surname> <given-names>F</given-names></name><name><surname>Champely</surname> <given-names>S</given-names></name><name><surname>Chessel</surname> <given-names>D</given-names></name><name><surname>Chhay</surname> <given-names>L</given-names></name><name><surname>Cummins</surname> <given-names>C</given-names></name><name><surname>Dewey</surname> <given-names>M</given-names></name><name><surname>Doran</surname> <given-names>HC</given-names></name><name><surname>Dray</surname> <given-names>S</given-names></name><name><surname>Dupont</surname> <given-names>C</given-names></name><name><surname>Eddelbuettel</surname> <given-names>D</given-names></name><name><surname>Enos</surname> <given-names>J</given-names></name><name><surname>Ekstrom</surname> <given-names>C</given-names></name><name><surname>Elff</surname> <given-names>M</given-names></name><name><surname>Farebrother</surname> <given-names>RW</given-names></name><name><surname>Fox</surname> <given-names>J</given-names></name><name><surname>Francois</surname> <given-names>R</given-names></name><name><surname>Friendly</surname> <given-names>M</given-names></name><name><surname>Galili</surname> <given-names>T</given-names></name><name><surname>Gamer</surname> <given-names>M</given-names></name><name><surname>Gastwirth</surname> <given-names>JL</given-names></name><name><surname>Gel</surname> <given-names>YR</given-names></name><name><surname>Gegzna</surname> <given-names>V</given-names></name><name><surname>Gross</surname> <given-names>J</given-names></name><name><surname>Grothendieck G</surname> <given-names>J</given-names></name><name><surname>Heiberger</surname> <given-names>R</given-names></name><name><surname>Hoehle</surname> <given-names>M</given-names></name><name><surname>Hoffmann</surname> <given-names>CW</given-names></name><name><surname>Hojsgaard</surname> <given-names>S</given-names></name><name><surname>Hothorn</surname> <given-names>T</given-names></name><name><surname>Huerzeler</surname> <given-names>M</given-names></name><name><surname>Hui</surname> <given-names>WW</given-names></name><name><surname>Hurd</surname> <given-names>P</given-names></name><name><surname>Hyndman</surname> <given-names>RJ</given-names></name><name><surname>Iglesias</surname> <given-names>PJV</given-names></name><name><surname>Jackson</surname> <given-names>C</given-names></name><name><surname>Kohl</surname> <given-names>M</given-names></name><name><surname>Korpela</surname> <given-names>M</given-names></name><name><surname>Kuhn</surname> <given-names>M</given-names></name><name><surname>Labes</surname> <given-names>D</given-names></name><name><surname>Lang</surname> <given-names>DT</given-names></name><name><surname>Leisch</surname> <given-names>F</given-names></name><name><surname>Lemon</surname> <given-names>J</given-names></name><name><surname>Li</surname> <given-names>D</given-names></name><name><surname>Maechler</surname> <given-names>M</given-names></name><name><surname>Magnusson</surname> <given-names>A</given-names></name><name><surname>Mainwaring</surname> <given-names>B</given-names></name><name><surname>Malter</surname> <given-names>D</given-names></name><name><surname>Marsaglia</surname> <given-names>G</given-names></name><name><surname>Marsaglia</surname> <given-names>J</given-names></name><name><surname>Matei</surname> <given-names>A</given-names></name><name><surname>Meyer</surname> <given-names>D</given-names></name><name><surname>Miao</surname> <given-names>W</given-names></name><name><surname>Millo</surname> <given-names>G</given-names></name><name><surname>Min</surname> <given-names>Y</given-names></name><name><surname>Mitchell</surname> <given-names>D</given-names></name><name><surname>Mueller</surname> <given-names>F</given-names></name><name><surname>Naepflin</surname> <given-names>M</given-names></name><name><surname>Navarro</surname> <given-names>D</given-names></name><name><surname>Nilsson</surname> <given-names>H</given-names></name><name><surname>Nordhausen</surname> <given-names>K</given-names></name><name><surname>Ogle</surname> <given-names>D</given-names></name><name><surname>Ooi</surname> <given-names>H</given-names></name><name><surname>Parsons</surname> <given-names>N</given-names></name><name><surname>Pavoine</surname> <given-names>S</given-names></name><name><surname>Plate</surname> <given-names>T</given-names></name><name><surname>Rapold</surname> <given-names>R</given-names></name><name><surname>Revelle</surname> <given-names>W</given-names></name><name><surname>Rinker</surname> <given-names>T</given-names></name><name><surname>Ripley</surname> <given-names>BD</given-names></name><name><surname>Rodriguez</surname> <given-names>C</given-names></name><name><surname>Russell</surname> <given-names>N</given-names></name><name><surname>Sabbe</surname> <given-names>N</given-names></name><name><surname>Seshan</surname> <given-names>VE</given-names></name><name><surname>Snow</surname> <given-names>G</given-names></name><name><surname>Smithson</surname> <given-names>M</given-names></name><name><surname>Soetaert</surname> <given-names>K</given-names></name><name><surname>Stahel</surname> <given-names>WA</given-names></name><name><surname>Stephenson</surname> <given-names>A</given-names></name><name><surname>Stevenson</surname> <given-names>M</given-names></name><name><surname>Stubner</surname> <given-names>R</given-names></name><name><surname>Templ</surname> <given-names>M</given-names></name><name><surname>Therneau</surname> <given-names>T</given-names></name><name><surname>Tille</surname> <given-names>Y</given-names></name><name><surname>Torgo</surname> <given-names>L</given-names></name><name><surname>Trapletti</surname> <given-names>A</given-names></name><name><surname>Ulrich</surname> <given-names>J</given-names></name><name><surname>Ushey</surname> <given-names>K</given-names></name><name><surname>VanDerWal</surname> <given-names>J</given-names></name><name><surname>Venables</surname> <given-names>B</given-names></name><name><surname>Verzani</surname> <given-names>J</given-names></name><name><surname>Warnes</surname> <given-names>GR</given-names></name><name><surname>Wellek</surname> <given-names>S</given-names></name><name><surname>Wickham</surname> <given-names>H</given-names></name><name><surname>Wilcox</surname> <given-names>RR</given-names></name><name><surname>Wolf</surname> <given-names>P</given-names></name><name><surname>Wollschlaeger</surname> <given-names>D</given-names></name><name><surname>Wood</surname> <given-names>J</given-names></name><name><surname>Wu</surname> <given-names>Y</given-names></name><name><surname>Yee</surname> <given-names>T</given-names></name><name><surname>Zeileis</surname> <given-names>A</given-names></name></person-group><year iso-8601-date="2020">2020</year><source>DescTools: Tools for Descriptive Statistics</source></element-citation></ref><ref id="bib99"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Sinha</surname> <given-names>H</given-names></name><name><surname>David</surname> <given-names>L</given-names></name><name><surname>Pascon</surname> <given-names>RC</given-names></name><name><surname>Clauder-Münster</surname> <given-names>S</given-names></name><name><surname>Krishnakumar</surname> <given-names>S</given-names></name><name><surname>Nguyen</surname> <given-names>M</given-names></name><name><surname>Shi</surname> <given-names>G</given-names></name><name><surname>Dean</surname> <given-names>J</given-names></name><name><surname>Davis</surname> <given-names>RW</given-names></name><name><surname>Oefner</surname> <given-names>PJ</given-names></name><name><surname>McCusker</surname> <given-names>JH</given-names></name><name><surname>Steinmetz</surname> <given-names>LM</given-names></name></person-group><year iso-8601-date="2008">2008</year><article-title>Sequential elimination of major-effect contributors identifies additional quantitative trait loci conditioning high-temperature growth in yeast</article-title><source>Genetics</source><volume>180</volume><fpage>1661</fpage><lpage>1670</lpage><pub-id pub-id-type="doi">10.1534/genetics.108.092932</pub-id><pub-id pub-id-type="pmid">18780730</pub-id></element-citation></ref><ref id="bib100"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Smith</surname> <given-names>RP</given-names></name><name><surname>Taher</surname> <given-names>L</given-names></name><name><surname>Patwardhan</surname> <given-names>RP</given-names></name><name><surname>Kim</surname> <given-names>MJ</given-names></name><name><surname>Inoue</surname> <given-names>F</given-names></name><name><surname>Shendure</surname> <given-names>J</given-names></name><name><surname>Ovcharenko</surname> <given-names>I</given-names></name><name><surname>Ahituv</surname> <given-names>N</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>Massively parallel decoding of mammalian regulatory sequences supports a flexible organizational model</article-title><source>Nature Genetics</source><volume>45</volume><fpage>1021</fpage><lpage>1028</lpage><pub-id pub-id-type="doi">10.1038/ng.2713</pub-id><pub-id pub-id-type="pmid">23892608</pub-id></element-citation></ref><ref id="bib101"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Spivak</surname> <given-names>AT</given-names></name><name><surname>Stormo</surname> <given-names>GD</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>ScerTF: a comprehensive database of benchmarked position weight matrices for Saccharomyces species</article-title><source>Nucleic Acids Research</source><volume>40</volume><fpage>D162</fpage><lpage>D168</lpage><pub-id pub-id-type="doi">10.1093/nar/gkr1180</pub-id><pub-id pub-id-type="pmid">22140105</pub-id></element-citation></ref><ref id="bib102"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Steinmetz</surname> <given-names>LM</given-names></name><name><surname>Sinha</surname> <given-names>H</given-names></name><name><surname>Richards</surname> <given-names>DR</given-names></name><name><surname>Spiegelman</surname> <given-names>JI</given-names></name><name><surname>Oefner</surname> <given-names>PJ</given-names></name><name><surname>McCusker</surname> <given-names>JH</given-names></name><name><surname>Davis</surname> <given-names>RW</given-names></name></person-group><year iso-8601-date="2002">2002</year><article-title>Dissecting the architecture of a quantitative trait locus in yeast</article-title><source>Nature</source><volume>416</volume><fpage>326</fpage><lpage>330</lpage><pub-id pub-id-type="doi">10.1038/416326a</pub-id><pub-id pub-id-type="pmid">11907579</pub-id></element-citation></ref><ref id="bib103"><element-citation publication-type="software"><person-group person-group-type="author"><name><surname>Storey</surname> <given-names>JD</given-names></name><name><surname>Bass</surname> <given-names>A</given-names></name><name><surname>Dabney</surname> <given-names>A</given-names></name><name><surname>Robinson</surname> <given-names>D</given-names></name></person-group><year iso-8601-date="2020">2020</year><source>Qvalue: Q-Value Estimation for False Discovery Rate Control</source></element-citation></ref><ref id="bib104"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Storey</surname> <given-names>JD</given-names></name><name><surname>Tibshirani</surname> <given-names>R</given-names></name></person-group><year iso-8601-date="2003">2003</year><article-title>Statistical significance for genomewide studies</article-title><source>PNAS</source><volume>100</volume><fpage>9440</fpage><lpage>9445</lpage><pub-id pub-id-type="doi">10.1073/pnas.1530509100</pub-id><pub-id pub-id-type="pmid">12883005</pub-id></element-citation></ref><ref id="bib105"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Stranger</surname> <given-names>BE</given-names></name><name><surname>Forrest</surname> <given-names>MS</given-names></name><name><surname>Clark</surname> <given-names>AG</given-names></name><name><surname>Minichiello</surname> <given-names>MJ</given-names></name><name><surname>Deutsch</surname> <given-names>S</given-names></name><name><surname>Lyle</surname> <given-names>R</given-names></name><name><surname>Hunt</surname> <given-names>S</given-names></name><name><surname>Kahl</surname> <given-names>B</given-names></name><name><surname>Antonarakis</surname> <given-names>SE</given-names></name><name><surname>Tavaré</surname> <given-names>S</given-names></name><name><surname>Deloukas</surname> <given-names>P</given-names></name><name><surname>Dermitzakis</surname> <given-names>ET</given-names></name></person-group><year iso-8601-date="2005">2005</year><article-title>Genome-Wide associations of gene expression variation in humans</article-title><source>PLOS Genetics</source><volume>1</volume><elocation-id>e78</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pgen.0010078</pub-id></element-citation></ref><ref id="bib106"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Tanay</surname> <given-names>A</given-names></name></person-group><year iso-8601-date="2006">2006</year><article-title>Extensive low-affinity transcriptional interactions in the yeast genome</article-title><source>Genome Research</source><volume>16</volume><fpage>962</fpage><lpage>972</lpage><pub-id pub-id-type="doi">10.1101/gr.5113606</pub-id><pub-id pub-id-type="pmid">16809671</pub-id></element-citation></ref><ref id="bib107"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Tewhey</surname> <given-names>R</given-names></name><name><surname>Kotliar</surname> <given-names>D</given-names></name><name><surname>Park</surname> <given-names>DS</given-names></name><name><surname>Liu</surname> <given-names>B</given-names></name><name><surname>Winnicki</surname> <given-names>S</given-names></name><name><surname>Reilly</surname> <given-names>SK</given-names></name><name><surname>Andersen</surname> <given-names>KG</given-names></name><name><surname>Mikkelsen</surname> <given-names>TS</given-names></name><name><surname>Lander</surname> <given-names>ES</given-names></name><name><surname>Schaffner</surname> <given-names>SF</given-names></name><name><surname>Sabeti</surname> <given-names>PC</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Direct identification of hundreds of Expression-Modulating variants using a multiplexed reporter assay</article-title><source>Cell</source><volume>165</volume><fpage>1519</fpage><lpage>1529</lpage><pub-id pub-id-type="doi">10.1016/j.cell.2016.04.027</pub-id><pub-id pub-id-type="pmid">27259153</pub-id></element-citation></ref><ref id="bib108"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ulirsch</surname> <given-names>JC</given-names></name><name><surname>Nandakumar</surname> <given-names>SK</given-names></name><name><surname>Wang</surname> <given-names>L</given-names></name><name><surname>Giani</surname> <given-names>FC</given-names></name><name><surname>Zhang</surname> <given-names>X</given-names></name><name><surname>Rogov</surname> <given-names>P</given-names></name><name><surname>Melnikov</surname> <given-names>A</given-names></name><name><surname>McDonel</surname> <given-names>P</given-names></name><name><surname>Do</surname> <given-names>R</given-names></name><name><surname>Mikkelsen</surname> <given-names>TS</given-names></name><name><surname>Sankaran</surname> <given-names>VG</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Systematic functional dissection of common genetic variation affecting red blood cell traits</article-title><source>Cell</source><volume>165</volume><fpage>1530</fpage><lpage>1545</lpage><pub-id pub-id-type="doi">10.1016/j.cell.2016.04.048</pub-id><pub-id pub-id-type="pmid">27259154</pub-id></element-citation></ref><ref id="bib109"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>van Arensbergen</surname> <given-names>J</given-names></name><name><surname>Pagie</surname> <given-names>L</given-names></name><name><surname>FitzPatrick</surname> <given-names>VD</given-names></name><name><surname>de Haas</surname> <given-names>M</given-names></name><name><surname>Baltissen</surname> <given-names>MP</given-names></name><name><surname>Comoglio</surname> <given-names>F</given-names></name><name><surname>van der Weide</surname> <given-names>RH</given-names></name><name><surname>Teunissen</surname> <given-names>H</given-names></name><name><surname>Võsa</surname> <given-names>U</given-names></name><name><surname>Franke</surname> <given-names>L</given-names></name><name><surname>de Wit</surname> <given-names>E</given-names></name><name><surname>Vermeulen</surname> <given-names>M</given-names></name><name><surname>Bussemaker</surname> <given-names>HJ</given-names></name><name><surname>van Steensel</surname> <given-names>B</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>High-throughput identification of human SNPs affecting regulatory element activity</article-title><source>Nature Genetics</source><volume>51</volume><fpage>1160</fpage><lpage>1169</lpage><pub-id pub-id-type="doi">10.1038/s41588-019-0455-2</pub-id><pub-id pub-id-type="pmid">31253979</pub-id></element-citation></ref><ref id="bib110"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Vockley</surname> <given-names>CM</given-names></name><name><surname>Guo</surname> <given-names>C</given-names></name><name><surname>Majoros</surname> <given-names>WH</given-names></name><name><surname>Nodzenski</surname> <given-names>M</given-names></name><name><surname>Scholtens</surname> <given-names>DM</given-names></name><name><surname>Hayes</surname> <given-names>MG</given-names></name><name><surname>Lowe</surname> <given-names>WL</given-names></name><name><surname>Reddy</surname> <given-names>TE</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>Massively parallel quantification of the regulatory effects of noncoding genetic variation in a human cohort</article-title><source>Genome Research</source><volume>25</volume><fpage>1206</fpage><lpage>1214</lpage><pub-id pub-id-type="doi">10.1101/gr.190090.115</pub-id><pub-id pub-id-type="pmid">26084464</pub-id></element-citation></ref><ref id="bib111"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>X</given-names></name><name><surname>He</surname> <given-names>L</given-names></name><name><surname>Goggin</surname> <given-names>SM</given-names></name><name><surname>Saadat</surname> <given-names>A</given-names></name><name><surname>Wang</surname> <given-names>L</given-names></name><name><surname>Sinnott-Armstrong</surname> <given-names>N</given-names></name><name><surname>Claussnitzer</surname> <given-names>M</given-names></name><name><surname>Kellis</surname> <given-names>M</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>High-resolution genome-wide functional dissection of transcriptional regulatory regions and nucleotides in human</article-title><source>Nature Communications</source><volume>9</volume><fpage>1</fpage><lpage>15</lpage><pub-id pub-id-type="doi">10.1038/s41467-018-07746-1</pub-id></element-citation></ref><ref id="bib112"><element-citation publication-type="software"><person-group person-group-type="author"><name><surname>Warnes</surname> <given-names>G</given-names></name></person-group><year iso-8601-date="2019">2019</year><source>Genetics: Population Genetics</source></element-citation></ref><ref id="bib113"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Weingarten-Gabbay</surname> <given-names>S</given-names></name><name><surname>Elias-Kirma</surname> <given-names>S</given-names></name><name><surname>Nir</surname> <given-names>R</given-names></name><name><surname>Gritsenko</surname> <given-names>AA</given-names></name><name><surname>Stern-Ginossar</surname> <given-names>N</given-names></name><name><surname>Yakhini</surname> <given-names>Z</given-names></name><name><surname>Weinberger</surname> <given-names>A</given-names></name><name><surname>Segal</surname> <given-names>E</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Comparative genetics systematic discovery of cap-independent translation sequences in human and viral genomes</article-title><source>Science</source><volume>351</volume><elocation-id>aad4939</elocation-id><pub-id pub-id-type="doi">10.1126/science.aad4939</pub-id><pub-id pub-id-type="pmid">26816383</pub-id></element-citation></ref><ref id="bib114"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Weingarten-Gabbay</surname> <given-names>S</given-names></name><name><surname>Nir</surname> <given-names>R</given-names></name><name><surname>Lubliner</surname> <given-names>S</given-names></name><name><surname>Sharon</surname> <given-names>E</given-names></name><name><surname>Kalma</surname> <given-names>Y</given-names></name><name><surname>Weinberger</surname> <given-names>A</given-names></name><name><surname>Segal</surname> <given-names>E</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Systematic interrogation of human promoters</article-title><source>Genome Research</source><volume>29</volume><fpage>171</fpage><lpage>183</lpage><pub-id pub-id-type="doi">10.1101/gr.236075.118</pub-id><pub-id pub-id-type="pmid">30622120</pub-id></element-citation></ref><ref id="bib115"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>West</surname> <given-names>MA</given-names></name><name><surname>Kim</surname> <given-names>K</given-names></name><name><surname>Kliebenstein</surname> <given-names>DJ</given-names></name><name><surname>van Leeuwen</surname> <given-names>H</given-names></name><name><surname>Michelmore</surname> <given-names>RW</given-names></name><name><surname>Doerge</surname> <given-names>RW</given-names></name><name><surname>St Clair</surname> <given-names>DA</given-names></name></person-group><year iso-8601-date="2007">2007</year><article-title>Global eQTL mapping reveals the complex genetic architecture of transcript-level variation in Arabidopsis</article-title><source>Genetics</source><volume>175</volume><fpage>1441</fpage><lpage>1450</lpage><pub-id pub-id-type="doi">10.1534/genetics.106.064972</pub-id><pub-id pub-id-type="pmid">17179097</pub-id></element-citation></ref><ref id="bib116"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Wickham</surname> <given-names>H</given-names></name><name><surname>Averick</surname> <given-names>M</given-names></name><name><surname>Bryan</surname> <given-names>J</given-names></name><name><surname>Chang</surname> <given-names>W</given-names></name><name><surname>McGowan</surname> <given-names>L</given-names></name><name><surname>François</surname> <given-names>R</given-names></name><name><surname>Grolemund</surname> <given-names>G</given-names></name><name><surname>Hayes</surname> <given-names>A</given-names></name><name><surname>Henry</surname> <given-names>L</given-names></name><name><surname>Hester</surname> <given-names>J</given-names></name><name><surname>Kuhn</surname> <given-names>M</given-names></name><name><surname>Pedersen</surname> <given-names>T</given-names></name><name><surname>Miller</surname> <given-names>E</given-names></name><name><surname>Bache</surname> <given-names>S</given-names></name><name><surname>Müller</surname> <given-names>K</given-names></name><name><surname>Ooms</surname> <given-names>J</given-names></name><name><surname>Robinson</surname> <given-names>D</given-names></name><name><surname>Seidel</surname> <given-names>D</given-names></name><name><surname>Spinu</surname> <given-names>V</given-names></name><name><surname>Takahashi</surname> <given-names>K</given-names></name><name><surname>Vaughan</surname> <given-names>D</given-names></name><name><surname>Wilke</surname> <given-names>C</given-names></name><name><surname>Woo</surname> <given-names>K</given-names></name><name><surname>Yutani</surname> <given-names>H</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Welcome to the tidyverse</article-title><source>Journal of Open Source Software</source><volume>4</volume><elocation-id>1686</elocation-id><pub-id pub-id-type="doi">10.21105/joss.01686</pub-id></element-citation></ref><ref id="bib117"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Wittkopp</surname> <given-names>PJ</given-names></name><name><surname>Haerum</surname> <given-names>BK</given-names></name><name><surname>Clark</surname> <given-names>AG</given-names></name></person-group><year iso-8601-date="2004">2004</year><article-title>Evolutionary changes in Cis and trans gene regulation</article-title><source>Nature</source><volume>430</volume><fpage>85</fpage><lpage>88</lpage><pub-id pub-id-type="doi">10.1038/nature02698</pub-id><pub-id pub-id-type="pmid">15229602</pub-id></element-citation></ref><ref id="bib118"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Wunderlich</surname> <given-names>Z</given-names></name><name><surname>Mirny</surname> <given-names>LA</given-names></name></person-group><year iso-8601-date="2009">2009</year><article-title>Different gene regulation strategies revealed by analysis of binding motifs</article-title><source>Trends in Genetics</source><volume>25</volume><fpage>434</fpage><lpage>440</lpage><pub-id pub-id-type="doi">10.1016/j.tig.2009.08.003</pub-id><pub-id pub-id-type="pmid">19815308</pub-id></element-citation></ref><ref id="bib119"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Yao</surname> <given-names>DW</given-names></name><name><surname>O'Connor</surname> <given-names>LJ</given-names></name><name><surname>Price</surname> <given-names>AL</given-names></name><name><surname>Gusev</surname> <given-names>A</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>Quantifying genetic effects on disease mediated by assayed gene expression levels</article-title><source>Nature Genetics</source><volume>52</volume><fpage>626</fpage><lpage>633</lpage><pub-id pub-id-type="doi">10.1038/s41588-020-0625-2</pub-id><pub-id pub-id-type="pmid">32424349</pub-id></element-citation></ref><ref id="bib120"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>J</given-names></name><name><surname>Kobert</surname> <given-names>K</given-names></name><name><surname>Flouri</surname> <given-names>T</given-names></name><name><surname>Stamatakis</surname> <given-names>A</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>PEAR: a fast and accurate illumina Paired-End reAd mergeR</article-title><source>Bioinformatics</source><volume>30</volume><fpage>614</fpage><lpage>620</lpage><pub-id pub-id-type="doi">10.1093/bioinformatics/btt593</pub-id><pub-id pub-id-type="pmid">24142950</pub-id></element-citation></ref><ref id="bib121"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Zheng</surname> <given-names>DQ</given-names></name><name><surname>Wang</surname> <given-names>PM</given-names></name><name><surname>Chen</surname> <given-names>J</given-names></name><name><surname>Zhang</surname> <given-names>K</given-names></name><name><surname>Liu</surname> <given-names>TZ</given-names></name><name><surname>Wu</surname> <given-names>XC</given-names></name><name><surname>Li</surname> <given-names>YD</given-names></name><name><surname>Zhao</surname> <given-names>YH</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>Genome sequencing and genetic breeding of a bioethanol <italic>Saccharomyces cerevisiae</italic> strain YJS329</article-title><source>BMC Genomics</source><volume>13</volume><elocation-id>479</elocation-id><pub-id pub-id-type="doi">10.1186/1471-2164-13-479</pub-id><pub-id pub-id-type="pmid">22978491</pub-id></element-citation></ref><ref id="bib122"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Zhou</surname> <given-names>J</given-names></name><name><surname>Theesfeld</surname> <given-names>CL</given-names></name><name><surname>Yao</surname> <given-names>K</given-names></name><name><surname>Chen</surname> <given-names>KM</given-names></name><name><surname>Wong</surname> <given-names>AK</given-names></name><name><surname>Troyanskaya</surname> <given-names>OG</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Deep learning sequence-based ab initio prediction of variant effects on expression and disease risk</article-title><source>Nature Genetics</source><volume>50</volume><fpage>1171</fpage><lpage>1179</lpage><pub-id pub-id-type="doi">10.1038/s41588-018-0160-6</pub-id><pub-id pub-id-type="pmid">30013180</pub-id></element-citation></ref><ref id="bib123"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Zhou</surname> <given-names>J</given-names></name><name><surname>Troyanskaya</surname> <given-names>OG</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>Predicting effects of noncoding variants with deep learning-based sequence model</article-title><source>Nature Methods</source><volume>12</volume><fpage>931</fpage><lpage>934</lpage><pub-id pub-id-type="doi">10.1038/nmeth.3547</pub-id><pub-id pub-id-type="pmid">26301843</pub-id></element-citation></ref><ref id="bib124"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Zhu</surname> <given-names>F</given-names></name><name><surname>Farnung</surname> <given-names>L</given-names></name><name><surname>Kaasinen</surname> <given-names>E</given-names></name><name><surname>Sahu</surname> <given-names>B</given-names></name><name><surname>Yin</surname> <given-names>Y</given-names></name><name><surname>Wei</surname> <given-names>B</given-names></name><name><surname>Dodonova</surname> <given-names>SO</given-names></name><name><surname>Nitta</surname> <given-names>KR</given-names></name><name><surname>Morgunova</surname> <given-names>E</given-names></name><name><surname>Taipale</surname> <given-names>M</given-names></name><name><surname>Cramer</surname> <given-names>P</given-names></name><name><surname>Taipale</surname> <given-names>J</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>The interaction landscape between transcription factors and the nucleosome</article-title><source>Nature</source><volume>562</volume><fpage>76</fpage><lpage>81</lpage><pub-id pub-id-type="doi">10.1038/s41586-018-0549-5</pub-id><pub-id pub-id-type="pmid">30250250</pub-id></element-citation></ref></ref-list></back><sub-article article-type="decision-letter" id="sa1"><front-stub><article-id pub-id-type="doi">10.7554/eLife.62669.sa1</article-id><title-group><article-title>Decision letter</article-title></title-group><contrib-group><contrib contrib-type="editor"><name><surname>Landry</surname><given-names>Christian R</given-names></name><role>Reviewing Editor</role><aff><institution>Université Laval</institution><country>Canada</country></aff></contrib></contrib-group></front-stub><body><boxed-text><p>In the interests of transparency, eLife publishes the most substantive revision requests and the accompanying author responses.</p></boxed-text><p><bold>Acceptance summary:</bold></p><p>The authors devised a new approach based on oligo synthesis, a transcriptional reporter and barcode sequencing to identify likely causal variants underlying cis regulatory variation in yeast. The results show that some promoter regions often have multiple SNPs affecting gene expression. The authors find that some of these regulatory SNPs show epistatic interactions and that natural selection may keep regulatory SNPs at low frequency in natural populations. SNPs affecting gene expression are enriched in known transcription factor binding sites. This study is a spectacular example of how the combination of emerging and established technologies can be exploited to gain a refined picture of genotype-phenotype maps and this, genome wide.</p><p><bold>Decision letter after peer review:</bold></p><p>Thank you for submitting your article &quot;Massively parallel identification of causal variants underlying gene expression differences in a yeast cross&quot; for consideration by <italic>eLife</italic>. Your article has been reviewed by three peer reviewers, and the evaluation has been overseen by a Reviewing Editor and Patricia Wittkopp as the Senior Editor. The reviewers have opted to remain anonymous.</p><p>The reviewers have discussed the reviews with one another and the Reviewing Editor has drafted this decision to help you prepare a revised submission.</p><p>We would like to draw your attention to changes in our revision policy that we have made in response to COVID-19 (https://elifesciences.org/articles/57162). Specifically, we are asking editors to accept without delay manuscripts, like yours, that they judge can stand as <italic>eLife</italic> papers without additional data, even if they feel that they would make the manuscript stronger. Thus the revisions requested below only address clarity and presentation.</p><p>Summary:</p><p>The authors devised a new approach based on oligo synthesis, a transcriptional reporter and barcode sequencing to identify causal variants underlying cis regulatory variation in yeast. The results show that some promoter regions often have multiple SNPs affecting gene expression. The authors find that some of these regulatory SNPs show epistatic interactions and that natural selection may keep regulatory SNPs at low frequency in natural populations. SNPs affecting gene expression are also enriched in known transcription factor binding sites.</p><p>The three reviews are highly positive. However, for the paper to be considered further, it would be important to perform some additional analyses and revise some of the interpretations of the results. This is particularly the case for some of the recommended analyses aimed at disentangling the relationships between different factors associating with regulatory SNPs because these often covary. If it is impossible to completely estimate their independent contribution, this issue should at least be addressed in the Results and Discussion.</p><p>I have kept the three full reports below because they are complementary and well detailed. We expect no additional experiments at this point.</p><p><italic>Reviewer #1:</italic></p><p>This study investigates the molecular origin of differential transcription in the well-studied BY-RM cross. SNPs and indels in promoters between these two strains were reciprocally exchanged in order to measure their effects by sequencing barcoded transcripts. The authors use this method to do a deep dive into the genetic determinants of local eQTLs, in which they identify causal variants and give examples of proximal variants with non-additive interactions. They also explore the nature of the causal variants and work to predict variants and expression. In my opinion, this article is very well written, well measured, well explained, and well thought through. I have no major comments.</p><p><italic>Reviewer #2:</italic></p><p>The paper by Renganaath et al. uses a reporter assay and Illumina sequencing to estimate the effects on gene expression of thousands of naturally occurring promoter variants between two yeast strains. They identify a large number of variants that have significant effects on expression, greatly expanding the catalog of known individual regulatory variants. They use this catalog to test long standing ideas about the molecular and evolutionary nature of these regulatory variants. Overall, the experiments use a good design with the necessary controls and replication to identify variants with moderate and large effects on gene expression. In general I think the work makes a good contribution to the field and I only have a few comments and concern about the models used and the strength of the conclusions made from these models.</p><p>1) First, the authors have a number of potential explanatory variables and test each one individually for association with whether a variant significantly alters gene expression or not. The results from these regression analyses are taken as is, with no attempt to account for correlations among the factors themselves. The authors seem to be aware of this issue; one of the largest individual correlates is gene essentiality which the authors note is often associated with some of the other covariates used. Because of this issue, significant associations cannot be interpreted as meaning a particular covariate is important, and causal connections can't be made from this kind of analysis. Furthermore, the authors take a number of significant covariates and interpret them as being consistent with negative selection, but whether these covariates are all significant once others are accounted for is unknown. The different covariates may be detecting similar signals, in which case they are not independent. A more comprehensive modeling scheme would allow the most important covariates to be identified and lead to a better understanding of what signals actually exist in the data. For example, using regularization techniques (which the authors do use later in another analysis) on a model including all of the covariates would help to avoid non-independent covariates.</p><p>2) Similar arguments can be made for the regression analyses that incorporate transcription factor binding; the PWMs of TFs are not independent and many have similarities due to similar modes of binding. In addition, the data does not clearly show 'evidence that causal variations often perturb TF binding'. Instead, the data shows that variants predicted to alter the binding of TFs are correlated with whether a variant is causal. Again, direct causality from this type of analysis is difficult to do. This is made even more difficult to follow with the “weak” TF binding as it appears that the authors are arguing for a model where individual single nucleotide variants affect expression by altering the binding of multiple TFs, each of which has a low individual probability of being bound. While it is well known that many weak TFBS are present in DNA, I'm not familiar with changes to these weak TFBS being proposed in the literature as a major route by which gene expression is altered in nature. Two things could help make this claim stronger. First, it would help to see a clear example that was found in the data, e.g. of a causal variant that was predicted to alter a single strong TFBS and one that was predicted to affect multiple weak TFBS. Second, functional validation of some of these variants as affecting the predicted TF binding (and not some other, unknown and untested factor) would significantly increase the impact of this section. This later route would likely require substantial effort, but at the very least, that such functional analysis has not been done needs to be recognized and the claims moderated in this section.</p><p><italic>Reviewer #3:</italic></p><p>Renganaath and colleagues described a high-throughput assay for testing how natural genetic variants influence gene expression, as well as finding patterns that allow us to predict such effects. The main novelty of their approach is in its single-variant resolution. Other methods such as eQTL mapping and allele-specific expression (ASE) do not have this level of resolution, so MPRA is a valuable addition to the yeast genetics toolkit. The ability to test epistatic interactions of neighboring variants is also a nice feature.</p><p>1) The role of individual variants in contributing to ASE is key to understanding cis-regulatory logic. The authors designed MPRA oligos focusing on previously identified ASE genes and added randomly selected genes as controls. It would be interesting to see an aggregated statistic on whether ASE genes are more likely to harbor causal variants than non-ASE genes; and if upstream variants differ from TSS variants in such patterns.</p><p>2) Prediction of causal regulatory variants is a holy grail of functional genomics. In this study, the predictions show slightly (but significantly) better than random performance. It is worth noting that there are more non-causal variants that causal variants, resulting in imbalanced classification problem. Oversampling of the minority labels could be a good start to balance data distribution and improve prediction performance.</p></body></sub-article><sub-article article-type="reply" id="sa2"><front-stub><article-id pub-id-type="doi">10.7554/eLife.62669.sa2</article-id><title-group><article-title>Author response</article-title></title-group></front-stub><body><disp-quote content-type="editor-comment"><p>Reviewer #2:</p><p>The paper by Renganaath et al. uses a reporter assay and Illumina sequencing to estimate the effects on gene expression of thousands of naturally occurring promoter variants between two yeast strains. They identify a large number of variants that have significant effects on expression, greatly expanding the catalog of known individual regulatory variants. They use this catalog to test long standing ideas about the molecular and evolutionary nature of these regulatory variants. Overall, the experiments use a good design with the necessary controls and replication to identify variants with moderate and large effects on gene expression. In general I think the work makes a good contribution to the field and I only have a few comments and concern about the models used and the strength of the conclusions made from these models.</p><p>1) First, the authors have a number of potential explanatory variables and test each one individually for association with whether a variant significantly alters gene expression or not. The results from these regression analyses are taken as is, with no attempt to account for correlations among the factors themselves. The authors seem to be aware of this issue; one of the largest individual correlates is gene essentiality which the authors note is often associated with some of the other covariates used. Because of this issue, significant associations cannot be interpreted as meaning a particular covariate is important, and causal connections can't be made from this kind of analysis. Furthermore, the authors take a number of significant covariates and interpret them as being consistent with negative selection, but whether these covariates are all significant once others are accounted for is unknown. The different covariates may be detecting similar signals, in which case they are not independent. A more comprehensive modeling scheme would allow the most important covariates to be identified and lead to a better understanding of what signals actually exist in the data. For example, using regularization techniques (which the authors do use later in another analysis) on a model including all of the covariates would help to avoid non-independent covariates.</p></disp-quote><p>We agree with the reviewer that many of our features are correlated and want to stress that in the single-feature analyses the reviewer refers to, we did not intend to identify individual features that contribute to variant causality independently of all other features. Instead, in our view, when multiple related features are associated with causality, this hints at the influence of an underlying phenomenon that may not be perfectly captured by any one feature alone. To stay with the example highlighted by the reviewer, the associations of variant causality with 1) lower derived allele frequency, 2) occurrence in the promoters of non-essential genes, and 3) occurrence in the promoters of genes with fewer synthetic genetic interactions, are all consistent with an underlying signal of negative selection against variants that alter the expression of essential genes.</p><p>To make this point obvious to the reader, we have added the following statement: “While these analyses cannot isolate the individual contributions of features that are correlated with each other, they provide an overview of the characteristics of causal variants.”</p><p>Following the reviewer’s suggestion of attempting to account for correlated features via regularization, we have extended our existing linear elastic net models (in which we had modeled variant effect size as a quantitative response variable) to logistic elastic net models of whether or not a variant is causal. Briefly, we fit all features in one model with 10-fold repeated cross-validation (5 repeats). This model predicted variant causality fairly poorly, with an area under the ROC of 0.53 on a 10% held-out test set. In this model, 1,812 predictors had a non-zero importance score. We also fit our 112 logistic regression models of various feature subsets with elastic net regularization. The best of these models, which included non-TF features and individual TF features in the strand agnostic configuration that had been significant in the univariate analyses, achieved an AUC-ROC of 0.71, identical to the best model without regularization. We have added these results to the paper (subsection “Prediction of causal variants”).</p><p>We draw two conclusions from these results. First, regularization alone does not improve predictions. We suspect that our dataset of a few hundred causal variants, each in a different promoter context, is not large enough to permit more accurate predictions. Second, and more directly to the reviewer’s concern, the feature associations we detected in our univariate logistic regressions do not trivially collapse to a small set of highly correlated features.</p><disp-quote content-type="editor-comment"><p>2) Similar arguments can be made for the regression analyses that incorporate transcription factor binding; the PWMs of TFs are not independent and many have similarities due to similar modes of binding. In addition, the data does not clearly show “evidence that causal variations often perturb TF binding”. Instead, the data shows that variants predicted to alter the binding of TFs are correlated with whether a variant is causal. Again, direct causality from this type of analysis is difficult to do.</p></disp-quote><p>We agree that our concluding statement for this section (“…provided clear evidence that causal variations often perturb TF binding”) was worded too strongly. Closely following the reviewer’s suggestion, we now conclude this section with “Overall, these analyses provided clear evidence that variants predicted to perturb TF binding are more likely to alter gene expression than variants not predicted to do so.”</p><p>As we pointed out in response to the reviewer’s first comment above, our intention in these analyses was not to assign independent contributions to binding of individual factors. As the reviewer points out, doing so would be extremely challenging due to sharing of motifs by different factors. To make this point clear for readers, we have added the statement “These analyses cannot disambiguate sharing of similar binding motifs by different TFs but probe the overall role of perturbed TF binding in variant causality.” Together with our reworded concluding sentence, we believe that this clarifies how we interpret the results on altered TFBSs.</p><disp-quote content-type="editor-comment"><p>This is made even more difficult to follow with the “weak” TF binding as it appears that the authors are arguing for a model where individual single nucleotide variants affect expression by altering the binding of multiple TFs, each of which has a low individual probability of being bound. While it is well known that many weak TFBS are present in DNA, I'm not familiar with changes to these weak TFBS being proposed in the literature as a major route by which gene expression is altered in nature. Two things could help make this claim stronger. First, it would help to see a clear example that was found in the data, e.g. of a causal variant that was predicted to alter a single strong TFBS and one that was predicted to affect multiple weak TFBS. Second, functional validation of some of these variants as affecting the predicted TF binding (and not some other, unknown and untested factor) would significantly increase the impact of this section. This later route would likely require substantial effort, but at the very least, that such functional analysis has not been done needs to be recognized and the claims moderated in this section.</p></disp-quote><p>The reviewer is correct that we argue for “a model where individual single nucleotide variants affect expression by altering the binding of multiple TFs, each of which has a low individual probability of being bound.”, although we do not strongly argue that such variants always have to alter binding of <italic>multiple</italic>​ TFs. Note that in the Discussion, we wrote “[variants] may perturb one or multiple weak binding sites”.​</p><p>This finding of strong association of causal variants with changes to weak TFBSs is supported by a recent study by de Boer et al., 2020, which we repeatedly cite in our Results and Discussion sections. The authors conducted a massively parallel reporter assay of more than 100 million random promoter fragments to build a highly predictive model of transcriptional regulation of RNA expression. In in-silico​ experiments probing their model, the authors set the level of individual TFs to zero and tested the influence of these “deletions” on gene expression. Strong regulatory effects (i.e., those that altered expression by more than 2-fold) of individual transcription factors were rare and accounted for only ~6% of the average expression driven by a typical promoter fragment. By contrast, the remaining 94% of the expression driven by a typical promoter fragment were attributed to the vast majority (99.9%) of interactions between a TF and a fragment that were individually weak. A key conclusion of the de Boer et al. paper is that (emphasis added) “[…] random DNA has diverse expression levels (Figure 1) that can be explained by TF binding (Figure 2), which regulate expression primarily through weak interactions​ (Figure ​ 6) that, in turn, can easily be perturbed​ [by changes to DNA sequence] (Figure 5).”. Here, we​ extend this reasoning by de Boer et al. to natural variants.</p><p>To help the reader understand these points about weak binding sites, we have added a new Figure 5—figure supplement 1, which shows an example of a variant with a clear change to a strong TFBS (panel A), as well as a variant that does not change any recognizable strong TFBSs, but does change predicted binding at multiple weak TFBSs (panel B).</p><p>We do acknowledge that additional work (which is outside of the scope of the current manuscript) will be required for a fuller understanding of the role of weak binding sites in regulatory variation. We have modified the Discussion section to make this point more clearly.</p><disp-quote content-type="editor-comment"><p>Reviewer #3:</p><p>Renganaath and colleagues described a high-throughput assay for testing how natural genetic variants influence gene expression, as well as finding patterns that allow us to predict such effects. The main novelty of their approach is in its single-variant resolution. Other methods such as eQTL mapping and allele-specific expression (ASE) do not have this level of resolution, so MPRA is a valuable addition to the yeast genetics toolkit. The ability to test epistatic interactions of neighboring variants is also a nice feature.</p><p>1) The role of individual variants in contributing to ASE is key to understanding cis-regulatory logic. The authors designed MPRA oligos focusing on previously identified ASE genes and added randomly selected genes as controls. It would be interesting to see an aggregated statistic on whether ASE genes are more likely to harbor causal variants than non-ASE genes; and if upstream variants differ from TSS variants in such patterns.</p></disp-quote><p>Thanks to the reviewer for this suggestion. We turned to ASE mRNA data previously gathered and integrated with local eQTL results in Albert et al.​​, 2018. Those data contain two independent BY/RM hybrid ASE datasets. Here, genes that had shown genome-wide significant ASE in at least one of these two datasets were more likely to have a causal MPRA variant (Fisher’s exact test: odds ratio = 3.2, p &lt; 2.2e-16). We have added this result to the Results section and a corresponding Materials and methods paragraph.</p><p>This agreement between ASE data and our MPRA assay is further bolstered by a correlation between the number of causal MPRA variants and either the number of ASE datasets with a significant result for the given gene (rho = 0.23, p &lt; 2.2e-16) or the absolute magnitude of ASE (rho = 0.14, p = 2e-8). These results were similar for variants in the TSS and Upstream MPRA libraries (<xref ref-type="table" rid="resptable1">Author response table 1</xref>).</p><table-wrap id="resptable1" position="anchor"><label>Author response table 1.</label><table frame="hsides" rules="groups"><thead><tr><th>Library</th><th>Fisher’s exact test (odds ratio)</th><th>Fisher’s exact test (p-value)</th><th>Correlation w/ number significant <break/>ASE <break/>datasets (rho)</th><th>Correlation w/ number significant <break/>ASE <break/>datasets (p-value)</th><th>Correlation with ASE magnitude <break/>(rho)</th><th>Correlation with ASE magnitude (p-value)</th></tr></thead><tbody><tr><td>TSS</td><td>3.1</td><td>3e-7</td><td>0.18</td><td>2e-8</td><td>0.11</td><td>0.0009</td></tr><tr><td>Upstream</td><td>2.8</td><td>6e-10</td><td>0.21</td><td>3e-12</td><td>0.13</td><td>0.00001</td></tr></tbody></table></table-wrap><p>In the interest of brevity, we chose not to include these more detailed analyses in the paper. Note that the ASE data we used in the analyses above differ slightly from those we had used for designing the MPRA. This is simply because more ASE data has become available since we designed the MPRA.</p><disp-quote content-type="editor-comment"><p>2) Prediction of causal regulatory variants is a holy grail of functional genomics. In this study, the predictions show slightly (but significantly) better than random performance. It is worth noting that there are more non-causal variants that causal variants, resulting in imbalanced classification problem. Oversampling of the minority labels could be a good start to balance data distribution and improve prediction performance.</p></disp-quote><p>We acknowledge the class imbalance in the training sets of our classifiers and thank the reviewer for this suggestion. To test if oversampling improves predictions, we oversampled causal variants to an equal representation of causal and non-causal variants in the training set used to train our classifiers. As expected, the classifiers with oversampling of the minority class were more stable during training and performed better in predicting on the <italic>validation</italic>​​ sets: across the 112 models, median average Cohen’s Kappa increased from 0.02 without oversampling to 0.22 with oversampling (see the top left panel in <xref ref-type="fig" rid="sa2fig1">Author response image 1</xref>; the red line denotes equality). There was a marginal improvement in the AUC-ROC when predicting on the <italic>test</italic>​ set (the best model improved from 0.71 to 0.73; see the top right panel in <xref ref-type="fig" rid="sa2fig1">Author response image 1</xref>; this best classifier used the same features in the training stage as in our analyses in the paper). However, the average performance across the 112 models remained similar (bottom panel; the red line corresponds to equal performance; note that the points are symmetric around this line indicating that there is no overall trend for better performance). Given this absence of a systematic improvement, we have chosen not to include these results in the paper.</p><fig id="sa2fig1"><label>Author response image 1.</label><graphic xlink:href="elife-62669-resp-fig1-v2.tif" mimetype="image" mime-subtype="tiff"/></fig></body></sub-article></article>