<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Archiving and Interchange DTD v1.1 20151215//EN"  "JATS-archivearticle1.dtd"><article article-type="research-article" dtd-version="1.1" xmlns:ali="http://www.niso.org/schemas/ali/1.0/" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink"><front><journal-meta><journal-id journal-id-type="nlm-ta">elife</journal-id><journal-id journal-id-type="publisher-id">eLife</journal-id><journal-title-group><journal-title>eLife</journal-title></journal-title-group><issn pub-type="epub" publication-format="electronic">2050-084X</issn><publisher><publisher-name>eLife Sciences Publications, Ltd</publisher-name></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">41279</article-id><article-id pub-id-type="doi">10.7554/eLife.41279</article-id><article-categories><subj-group subj-group-type="display-channel"><subject>Research Article</subject></subj-group><subj-group subj-group-type="heading"><subject>Computational and Systems Biology</subject></subj-group></article-categories><title-group><article-title>Synthetic and genomic regulatory elements reveal aspects of <italic>cis</italic>-regulatory grammar in mouse embryonic stem cells</article-title></title-group><contrib-group><contrib contrib-type="author" id="author-118529"><name><surname>King</surname><given-names>Dana M</given-names></name><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0003-4635-5272</contrib-id><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="fn" rid="con1"/><xref ref-type="fn" rid="conf1"/><xref ref-type="fn" rid="pa1">†</xref></contrib><contrib contrib-type="author" id="author-167650"><name><surname>Hong</surname><given-names>Clarice Kit Yee</given-names></name><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0002-9485-1425</contrib-id><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="fn" rid="con2"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author" id="author-167651"><name><surname>Shepherdson</surname><given-names>James L</given-names></name><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0002-3288-7000</contrib-id><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="fn" rid="con3"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author" id="author-167784"><name><surname>Granas</surname><given-names>David M</given-names></name><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="fn" rid="con4"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author" id="author-33540"><name><surname>Maricque</surname><given-names>Brett B</given-names></name><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="fn" rid="con5"/><xref ref-type="fn" rid="conf1"/><xref ref-type="fn" rid="pa2">‡</xref></contrib><contrib contrib-type="author" corresp="yes" id="author-13668"><name><surname>Cohen</surname><given-names>Barak A</given-names></name><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0002-3350-2715</contrib-id><email>cohen@wustl.edu</email><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="other" rid="fund1"/><xref ref-type="fn" rid="con6"/><xref ref-type="fn" rid="conf1"/></contrib><aff id="aff1"><label>1</label><institution>Edison Center for Genome Sciences and Systems Biology, Washington University in St. Louis</institution><addr-line><named-content content-type="city">St. Louis</named-content></addr-line><country>United States</country></aff><aff id="aff2"><label>2</label><institution>Department of Genetics, Washington University in St. Louis</institution><addr-line><named-content content-type="city">St. Louis</named-content></addr-line><country>United States</country></aff></contrib-group><contrib-group content-type="section"><contrib contrib-type="editor"><name><surname>Wittkopp</surname><given-names>Patricia J</given-names></name><role>Reviewing Editor</role><aff><institution>University of Michigan</institution><country>United States</country></aff></contrib><contrib contrib-type="senior_editor"><name><surname>Wittkopp</surname><given-names>Patricia J</given-names></name><role>Senior Editor</role><aff><institution>University of Michigan</institution><country>United States</country></aff></contrib></contrib-group><author-notes><fn fn-type="present-address" id="pa1"><label>†</label><p>University of Michigan, Bioinformatics Core, Ann Arbor, United States</p></fn><fn fn-type="present-address" id="pa2"><label>‡</label><p>Columbia University, Department of Psychology, New York, United States</p></fn></author-notes><pub-date date-type="publication" publication-format="electronic"><day>11</day><month>02</month><year>2020</year></pub-date><pub-date pub-type="collection"><year>2020</year></pub-date><volume>9</volume><elocation-id>e41279</elocation-id><history><date date-type="received" iso-8601-date="2018-08-20"><day>20</day><month>08</month><year>2018</year></date><date date-type="accepted" iso-8601-date="2020-02-07"><day>07</day><month>02</month><year>2020</year></date></history><permissions><copyright-statement>© 2020, King et al</copyright-statement><copyright-year>2020</copyright-year><copyright-holder>King et al</copyright-holder><ali:free_to_read/><license xlink:href="http://creativecommons.org/licenses/by/4.0/"><ali:license_ref>http://creativecommons.org/licenses/by/4.0/</ali:license_ref><license-p>This article is distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="http://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution License</ext-link>, which permits unrestricted use and redistribution provided that the original author and source are credited.</license-p></license></permissions><self-uri content-type="pdf" xlink:href="elife-41279-v2.pdf"/><abstract><p>In embryonic stem cells (ESCs), a core transcription factor (TF) network establishes the gene expression program necessary for pluripotency. To address how interactions between four key TFs contribute to <italic>cis-</italic>regulation in mouse ESCs, we assayed two massively parallel reporter assay (MPRA) libraries composed of binding sites for SOX2, POU5F1 (OCT4), KLF4, and ESRRB. Comparisons between synthetic <italic>cis</italic>-regulatory elements and genomic sequences with comparable binding site configurations revealed some aspects of a regulatory grammar. The expression of synthetic elements is influenced by both the number and arrangement of binding sites. This grammar plays only a small role for genomic sequences, as the relative activities of genomic sequences are best explained by the predicted occupancy of binding sites, regardless of binding site identity and positioning. Our results suggest that the effects of transcription factor binding sites (TFBS) are influenced by the order and orientation of sites, but that in the genome the overall occupancy of TFs is the primary determinant of activity.</p></abstract><abstract abstract-type="executive-summary"><title>eLife digest</title><p>Transcription factors are proteins that flip genetic switches; their role is to control when and where genes are active. They do this by binding to short stretches of DNA called <italic>cis</italic>-regulatory sequences. Each sequence can have several binding sites for different transcription factors, but it is largely unclear whether the transcription factors binding to the same regulatory sequence actually work together.</p><p>It is possible that each transcription factor may work independently and there only needs to be critical mass of transcription factors bound to throw the genetic switch. If this is the case, the most important features of a <italic>cis</italic>-regulatory sequence should be the number of binding sites it contains, and how tightly the transcription factors bind to those sites. The more transcription factors and the more strongly they bind, the more active the gene should be. An alternative option is that certain transcription factors may work better together, enhancing each other's effects such that the total effect is more than the sum of its parts. If this is true, the order, orientation and spacing of the binding sites within a sequence should matter more than the number.</p><p>One way to investigate to distinguish between these possibilities is to study mouse embryonic stem cells, which have a core set of four transcription factors. Looking directly at a real genome, however, can be confusing and it is difficult to measure the effects of different <italic>cis</italic>-regulatory sequences because genes differ in so many other ways. To tackle this problem, King et al. created a synthetic set of <italic>cis</italic>-regulatory sequences based on the four core transcription factors found in mouse stem cells.</p><p>The synthetic set had every combination of two, three or four of the binding sites, with each site either facing forwards or backwards along the DNA strand. King et al. attached each of the synthetic cis-regulatory sequences to a reporter gene to find out how well each sequence performed. This revealed that the cis-regulatory sequences with the most binding sites and the tightest binding affinities work best, suggesting that transcription factors mainly work independently.</p><p>There was evidence of some interaction between some transcription factors, because, of the synthetic sequences with four binding sites, some worked better than others, and there were patterns in the most effective binding site combinations. However, these effects were small and when King et al. went on to test sequences from the real mouse genome, the most important factor by far was the number of binding sites.</p><p>Synthetic libraries of DNA sequences allow researchers to examine gene regulation more clearly than is possible in real genomes. Yet this approach does have its limitations and it is impossible to capture every type of <italic>cis</italic>-regulatory sequence in one library. The next step to extend this work is to combine the two approaches, taking sequences from the real genome and manipulating them one by one. This could help to unravel the rules that govern how <italic>cis</italic>-regulatory sequences work in real cells.</p></abstract><kwd-group kwd-group-type="author-keywords"><kwd>gene expression</kwd><kwd>transcription factors</kwd><kwd>systems biology</kwd><kwd>pluripotency</kwd></kwd-group><kwd-group kwd-group-type="research-organism"><title>Research organism</title><kwd>Mouse</kwd></kwd-group><funding-group><award-group id="fund1"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100000002</institution-id><institution>National Institutes of Health</institution></institution-wrap></funding-source><award-id>R01 GM092910</award-id><principal-award-recipient><name><surname>Cohen</surname><given-names>Barak</given-names></name></principal-award-recipient></award-group><funding-statement>The funders had no role in study design, data collection and interpretation, or the decision to submit the work for publication.</funding-statement></funding-group><custom-meta-group><custom-meta specific-use="meta-only"><meta-name>Author impact statement</meta-name><meta-value>The independent effects of transcription factor binding sites are large regardless of sequence context, but the interactions between sites are context dependent.</meta-value></custom-meta></custom-meta-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><sec id="s1-1"><title>Independence versus interaction of transcription factor binding sites</title><p>Enhancers are composed of combinations of transcription factor binding sites (TFBS). An important question is: to what extent do TFBS act independently within enhancers and to what extent do specific interactions between transcription factors (TF) underlie enhancer function? Independence suggests a modular genome in which the effects of multiple binding sites are predictable from their individual effects. Interactions, such as cooperativity between TFs, cause the effect of multiple TFBS to be more (or less) than the combination of their individual effects. Constructing models that predict the expression of genes based on the TFBS composition of their surrounding regulatory DNA will require understanding the degree to which sites function independently and how interactions between sites contribute to the activity of regulatory sequences.</p></sec><sec id="s1-2"><title>Regulatory grammar</title><p>The extent to which TFs function either independently or through interactions should be reflected in the <italic>cis</italic>-regulatory <italic>grammar</italic> of TFBS, defined as the ways that the order, orientation, spacing, and affinity of binding sites impact the activity of enhancers. If TFs function independently then we do not expect strong constraints on the positioning of their binding sites within regulatory elements. If TFs function mostly through interactions with other TFs that require a precise geometry, then we expect strong biases in the positioning of TFBS within regulatory elements. At least three models make predictions of how grammar might influence enhancer activity, the billboard model, the enhanceosome model, and the TF collective model (<xref ref-type="bibr" rid="bib31">Kulkarni and Arnosti, 2003</xref>; <xref ref-type="bibr" rid="bib49">Spitz and Furlong, 2012</xref>). The enhanceosome model posits extensive interactions between bound TFs, resulting in a strict grammar in which only precise positioning of TFBS activate target genes. The enhanceosome model is supported by structural studies of the IFN-β enhancer, where a specific order and spacing of TFBS is required to activate expression (<xref ref-type="bibr" rid="bib42">Panne, 2008</xref>; <xref ref-type="bibr" rid="bib61">Yie et al., 1999</xref>). In contrast, the billboard model posits a more flexible grammar, where enhancers tolerate changes to the order, spacing, or orientations of TFBS with little change to target gene expression (<xref ref-type="bibr" rid="bib20">Giorgetti et al., 2010</xref>; <xref ref-type="bibr" rid="bib31">Kulkarni and Arnosti, 2003</xref>). In the billboard model bound TFs function in a largely independent manner. This model was proposed to explain binding site turnover in developmental enhancers and functional conservation of enhancer activity between species despite sequence divergence (<xref ref-type="bibr" rid="bib23">Hare et al., 2008a</xref>; <xref ref-type="bibr" rid="bib24">Hare et al., 2008b</xref>; <xref ref-type="bibr" rid="bib35">Ludwig et al., 2000</xref>; <xref ref-type="bibr" rid="bib53">Visel et al., 2009</xref>). In the TF collective model, specific TFs must be recruited to enhancers but can be recruited either by direct contact with DNA or indirectly through other TFs (<xref ref-type="bibr" rid="bib28">Junion et al., 2012</xref>; <xref ref-type="bibr" rid="bib49">Spitz and Furlong, 2012</xref>; <xref ref-type="bibr" rid="bib51">Uhl et al., 2016</xref>). In the collective model no specific TFBS is required for activity even though the recruitment of individual TFs might be. TFs may function independently in some contexts and may engage in interactions in other contexts. The billboard, enhanceosome, and collective models differ in the importance the precise arrangements of TFBS play in setting the activities of enhancers, and control of gene expression likely incorporates aspects of all three models. Quantifying the extent to which grammar influences activity in different contexts is an important step toward producing more predictive models of gene expression.</p><p>We and others have used mouse embryonic stem cells (mESCs) as a system for studying <italic>cis</italic>-regulatory grammar and cooperative interactions between the pluripotency factors POU5F1 (OCT4), SOX2, ESRRB, and KLF4 (<xref ref-type="bibr" rid="bib10">Dunn et al., 2014</xref>; <xref ref-type="bibr" rid="bib14">Fiore and Cohen, 2016</xref>; <xref ref-type="bibr" rid="bib59">Williams et al., 2004</xref>). The pluripotency factors are a core set of TFs that maintain pluripotency in mESCs and are sufficient to induce pluripotency in terminally differentiated cells (<xref ref-type="bibr" rid="bib13">Feng et al., 2009</xref>; <xref ref-type="bibr" rid="bib33">Liu et al., 2008</xref>; <xref ref-type="bibr" rid="bib40">Niwa, 2014</xref>; <xref ref-type="bibr" rid="bib50">Takahashi and Yamanaka, 2006</xref>; <xref ref-type="bibr" rid="bib62">Zhang et al., 2008</xref>). The pluripotency TFs activate self-renewal genes and repress genes that promote differentiation (<xref ref-type="bibr" rid="bib3">Chambers and Tomlinson, 2009</xref>). Based on known physical and genetic interactions, as well as genome-wide binding assays, multiple interacting TFs specify target gene expression in mESCs (<xref ref-type="bibr" rid="bib25">Huang et al., 2009</xref>; <xref ref-type="bibr" rid="bib40">Niwa, 2014</xref>; <xref ref-type="bibr" rid="bib45">Reményi et al., 2004</xref>; <xref ref-type="bibr" rid="bib44">Reményi et al., 2003</xref>; <xref ref-type="bibr" rid="bib59">Williams et al., 2004</xref>). However, it remains unclear how pluripotency TFs collaborate to drive-specific patterns of gene expression in ESCs, and what role, if any, is played by TFBS grammar in determining specificity in the genome (<xref ref-type="bibr" rid="bib3">Chambers and Tomlinson, 2009</xref>; <xref ref-type="bibr" rid="bib6">Chen et al., 2008b</xref>). Understanding how these factors combine to regulate their target genes is central to understanding the establishment and maintenance of the pluripotent state.</p><p>We previously addressed these questions by assaying a set of synthetic <italic>cis</italic>-regulatory elements that represent a small fraction of the possible arrangements of pluripotency TFBS. We identified some evidence for a grammar that is constrained by TFBS arrangement, including OCT4-SOX2 interactions. However, our previous study lacked sufficient power to detect other interactions (<xref ref-type="bibr" rid="bib14">Fiore and Cohen, 2016</xref>). Here, we explore the role of grammar for pluripotency TFBS by assaying an exhaustive set of synthetic <italic>cis</italic>-regulatory elements, composed of TFBS for SOX2, OCT4, KLF4 and ESRRB, as well as a limited set of genomic regulatory sequences with comparable configurations of binding sites. The pattern of expression of synthetic regulatory elements is well predicted by a model that incorporates binding site position. However, despite all genomic sequences overlapping ChIP-seq peaks for at least one of the four pluripotency factors, only about a third of sequences drove reporter gene activity above background levels. Additionally, the positional grammar learned from synthetic sequences performed poorly in predicting the activity of genomic sequences. Genomic sequences appear to also include sequence features that recruit additional TFs, either directly through TF-DNA interactions or possibly indirectly through TF-TF interactions. Our results suggest that in the genome the overall occupancy of TFs is the best predictor of binding site activity. Our results with synthetic elements suggest that other aspects of grammar (order, orientation) can tune the activity of sites, but these effects are difficult to observe without direct experimental manipulations. In the genome only the number and affinity of sites shows a correlation with activity.</p></sec></sec><sec id="s2" sec-type="results"><title>Results</title><sec id="s2-1"><title>Rationale and description of enhancer libraries</title><p>We designed two reporter gene libraries to explore the role of grammar in regulatory elements controlled by the pluripotency TFs. The first library, synthetic (SYN), contains a set of synthetic combinations of consensus TFBS for OCT4 (O), SOX2 (S), KLF4 (K), and ESRRB (E). We did not include sites for NANOG in our libraries as its position weight matrix (PWM) has low information content and is not amenable to a synthetic binding site approach. Nanog also appears to be dispensable for reprogramming terminal cells to a pluripotent state (<xref ref-type="bibr" rid="bib55">Wang et al., 2013</xref>; <xref ref-type="bibr" rid="bib54">Wang et al., 2012</xref>; <xref ref-type="bibr" rid="bib27">Jauch et al., 2008</xref>; <xref ref-type="bibr" rid="bib41">Pan and Thomson, 2007</xref>; <xref ref-type="bibr" rid="bib50">Takahashi and Yamanaka, 2006</xref>). We did not incorporate MYC-binding sites in our libraries because MYC often acts independently of the core pluripotency TFs (<xref ref-type="bibr" rid="bib8">Chen et al., 2012</xref>; <xref ref-type="bibr" rid="bib7">Chen et al., 2008c</xref>; <xref ref-type="bibr" rid="bib33">Liu et al., 2008</xref>).</p><p>We designed the SYN library to test how interactions between different TFs (heterotypic interactions) determine the activities of regulatory elements. If heterotypic interactions depend on the geometry of TF binding, then the order, orientation, and spacing of sites should influence activity. To test this prediction, we designed the SYN library to assay different orders and orientations of the pluripotency binding sites. The SYN library includes all possible 624 unique combinations of two, three, and four TFBS (2-mers, 3-mers, and 4-mers, respectively), with each TFBS in either the forward or reverse direction (<xref ref-type="supplementary-material" rid="supp1">Supplementary file 1A</xref>). Each synthetic element in the SYN library contains no more than one copy of a given TFBS. We chose this library design to focus on heterotypic interactions and to avoid the confounding effects of homotypic interactions, which we examined in detail in a previous study (<xref ref-type="bibr" rid="bib14">Fiore and Cohen, 2016</xref>). We embedded each TFBS in a constant 20 bp sequence with fixed spacing between sites to ensure that all the sites sit on the same side of the DNA helix. We avoided varying the length of the spacer sequence between sites because increasing the length of spacer sequences risks introducing cryptic binding sites that confound the results. For each TF, we used a consensus binding site based on its position weight matrix (PWM) in the JASPAR database (<xref ref-type="bibr" rid="bib46">Sandelin, 2004</xref>; <xref ref-type="bibr" rid="bib14">Fiore and Cohen, 2016</xref>). We did not vary the predicted affinity of the sites in the SYN library because we could not assay a library large enough to vary the affinity of sites while still testing all possible arrangements of sites. Our rationale was to retain the maximum power to detect the effects of the order and orientation of sites, and this required us to compromise on our ability to detect the effects of the spacing and affinity of sites. The highly controlled nature of the SYN library provides maximum power to detect interactions mediated by the order and orientation of sites.</p><p>The second library includes sequences from the mouse genome that match, as best as possible, members of the SYN library. Using the same PWMs used to design the SYN library, we scanned the mouse genome for combinations of the TFBS for O, S, K, and E within 100 bp of regions bound by any of the four pluripotency TFs in E14 mESCs as measured by ChIP-seq (<xref ref-type="bibr" rid="bib14">Fiore and Cohen, 2016</xref>; <xref ref-type="bibr" rid="bib1">Bailey et al., 2009</xref>; <xref ref-type="bibr" rid="bib7">Chen et al., 2008c</xref>). We chose genomic sequences that contain one and only one binding site that scores above the PWM threshold for each factor to mimic the composition of the SYN library. We identified few clusters that included all four binding sites (&lt;70). We therefore selected 407 genomic sequences with three pluripotency TFBS that could be compared to the exhaustive set of synthetic 3-mer elements. The resulting genomic wild-type library (gWT) is composed of 407 unique genomic sequences with combinations of any three of the four TFBS, with each site represented no more than once per sequence (Materials and methods, <xref ref-type="supplementary-material" rid="supp1">Supplementary file 1E-F</xref>). Although these sequences differ from SYN elements in the individual site affinities, spacings between TFBS, as well as intervening sequence composition, our expectation was that the gWT sequences would test how well interactions learned from the SYN library apply to genomic sequences. To confirm that the activity of the gWT sequences depends on the presence of pluripotency TFBS, we generated matched genomic mutant sequences (gMUT) in which all three of the identified pluripotency TFBS were mutated by changing two positions in each TFBS from the highest information content base to the lowest information base according to the PWM (<xref ref-type="fig" rid="fig1s1">Figure 1—figure supplement 1</xref>). The final gMUT sequences lack detectable TFBS for O, S, K, or E when rescanned with the threshold used to select the gWT sequences. The combined gWT/gMUT library allows us to quantify the contributions of the pluripotency sites to regulatory activity, as well as sample configurations of pluripotency TFBS from the genome that may provide insight into grammar for these sequences.</p></sec><sec id="s2-2"><title>MPRA of reporter gene libraries</title><p>We assayed the <italic>cis</italic>-regulatory activity of the SYN and gWT/gMUT libraries in mESCs using a plasmid-based Massively Parallel Reporter Assay (MPRA) (<xref ref-type="bibr" rid="bib32">Kwasnieski et al., 2012</xref>). Each unique library member described above is present eight times with a different unique sequence barcode (BC) in its 3’ UTR (<xref ref-type="bibr" rid="bib14">Fiore and Cohen, 2016</xref>). The elements were placed directly upstream of a minimal promoter, mirroring classical tests of enhancer activity. The assay does not, however, test whether elements can function as long-range enhancers. To determine the relative activity of each sequence compared to the minimal promoter included in each construct, we included copies of plasmids with only the minimal promoter paired with over a hundred unique BCs in each library (Materials and methods). Our measurements were highly reproducible between biological replicates, with R<sup>2</sup> between 0.98 and 0.99 for replicates of the SYN library and 0.96–0.98 for the gWT/gMUT library, and are not driven by abundance biases in the library (<xref ref-type="fig" rid="fig1s2">Figure 1—figure supplement 2</xref>). After thresholding on DNA and RNA counts, we recovered reads for 100% (624/624) of our SYN elements and 99% (403/407) of paired gWT/gMUT sequences. The high concordance between replicates and simultaneous sequencing of the two libraries allowed us to make quantitative comparisons, both within and between libraries.</p></sec><sec id="s2-3"><title>Synthetic and genomic libraries support different grammar models</title><p>TFBS in synthetic regulatory elements make strong independent contributions to expression. Most synthetic elements drive expression over basal activity regardless of the number, order, or orientation of sites within the element (<xref ref-type="fig" rid="fig1">Figure 1A</xref>). Of all SYN elements, 77% (6% of 2-mers, 66% of 3-mers, 92% of 4-mers) were statistically different from basal levels in all three replicates after correcting for multiple hypothesis testing (Wilcoxon rank-sum test; Bonferroni correction, n = 637; p-values reported in <xref ref-type="supplementary-material" rid="supp1">Supplementary file 1C</xref>). In most cases, three or four consensus binding sites are sufficient to increase expression above basal levels, which suggests strong independent contributions of TFBS to the activity of synthetic elements. Synthetic elements with more binding sites generally drive higher expression than elements with fewer binding sites, supporting the idea that TFBS can contribute to expression in an independent and additive manner. However, the wide range of expression levels observed from different 4-mer elements must be due to the arrangement of the TFBS, as site number, identity, and affinity are fixed. The strong positive effect of adding sites demonstrates an independent effect of TFBS, while the diversity of expression among elements with the same number of sites reveals that grammar can quantitatively modulate activity.</p><fig-group><fig id="fig1" position="float"><label>Figure 1.</label><caption><title>Activity of synthetic elements and genomic sequences.</title><p>(<bold>A</bold>) The activity of synthetic elements with different numbers of binding sites. Expression is the average log of the ratio of cDNA barcode counts/DNA barcode counts for each synthetic element normalized to basal expression (dotted line). (<bold>B</bold>) The activity of genomic sequences is largely dependent on the presence of pluripotency binding sites. Normalized expression of wild type (gWT) sequences is plotted against expression of matched sequences with all three pluripotency TFBS mutated (gMUT sequences). Red indicates sequences with significantly different expression between matched gWT and gMUT sequences. The diagonal solid line is the expectation if mutation of TFBS had no impact on expression level. Expression of both gWT and gMUT sequences are normalized to basal controls, but basal expression is only plotted for gWT sequences on the y-axis (dotted line).</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-41279-fig1-v2.tif"/></fig><fig id="fig1s1" position="float" specific-use="child-fig"><label>Figure 1—figure supplement 1.</label><caption><title>Pluripotency motif substitutions for gMUT sequences.</title><p>Highest information content positions in each motif were substituted with least frequent nucleotide for that position. (<bold>A</bold>) For mutating Sox2 motifs, the reference nucleotides were substituted for ‘A’ in position 4 and 5. (<bold>B</bold>) For mutating Oct4 motifs, the reference nucleotide was substituted for ‘C’ in position 2 and for ‘A’ in position 3. (<bold>C</bold>) For mutating Esrrb motifs, the reference nucleotide was substituted for ‘C’ in position five and ‘A’ for position 7. (<bold>D</bold>) For mutating Klf4 motifs, the reference nucleotide was substituted for ‘A’ in position three and ‘C’ in position 5.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-41279-fig1-figsupp1-v2.tif"/></fig><fig id="fig1s2" position="float" specific-use="child-fig"><label>Figure 1—figure supplement 2.</label><caption><title>MPRA data quality.</title><p>Reproducibility of barcode (BC) counts between biological replicates, normalized as reads per million per RNA replicate for (<bold>A</bold>) Synthetic library and (<bold>B</bold>) Genomic, gWT and gMUT, library. Comparison of normalized BC expression (BC<sub>RNA</sub>/BC<sub>DNA</sub>) versus DNA counts for (<bold>C</bold>) Synthetic library and (<bold>D</bold>) Genomic, gWT and gMUT, library.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-41279-fig1-figsupp2-v2.tif"/></fig></fig-group><p>In contrast to the synthetic elements, most genomic sequences in the gWT library did not exhibit regulatory activity above basal levels. Only 28% (113/403) of wild type genomic sequences were statistically different from basal levels in all three replicates (p&lt;0.05, Wilcoxon rank-sum test; Bonferroni correction, n = 403; p-values reported in <xref ref-type="supplementary-material" rid="supp1">Supplementary file 1H</xref>). This low fraction of active gWT sequences is consistent with observations from functional tests of genomic sequences bound by key TFs in other cell types (<xref ref-type="bibr" rid="bib15">Fisher et al., 2012</xref>; <xref ref-type="bibr" rid="bib22">Grossman et al., 2017</xref>; <xref ref-type="bibr" rid="bib57">White et al., 2013</xref>). The difference between the SYN and gWT libraries is that the surrounding sequence context in which the pluripotency sites occur in the gWT library varies much more than in the SYN library, and these contextual differences appear to have strong effects on the pluripotency sites. In most cases, the effect of sequence context in the gWT library was strong enough to suppress the independent contributions of the binding sites to activity. For genomic sequences that were statistically different from basal, 99% (112/113) have a significant difference between matched gWT and gMUT sequences (<xref ref-type="fig" rid="fig1">Figure 1B</xref>; p&lt;0.05, Wilcoxon rank-sum test; Bonferroni correction, n = 403; p-values reported in <xref ref-type="supplementary-material" rid="supp1">Supplementary file 1H</xref>), indicating that the activity of these genomic sequences depends on one or more of the pluripotency TFBS. Our observation that the presence of high-quality pluripotency TFBS is generally insufficient to drive expression demonstrates that binding sites must be presented in the proper surrounding sequence context in order to generate a functional regulatory element.</p></sec><sec id="s2-4"><title>Synthetic elements support a positional grammar</title><p>While the overall pattern of expression of SYN elements supports strong independent contributions from binding sites, direct comparisons of different TFBS configurations also support a role for interactions between factors. Pairwise comparisons between 3-mers and their matched 4-mers that include one additional site at either the 5’ or 3’ end, reveal that the position of the extra site can strongly influence expression. For example, the O-K-E 3-mer and the matched O-K-E-<underline>S</underline> 4-mer drive indistinguishable expression, while the matched <underline>S</underline>-O-K-E 4-mer drives one of the highest expression levels in the SYN library (<xref ref-type="fig" rid="fig2">Figure 2A</xref>). Other examples are consistent with either strong position dependence or both position and orientation dependence (<xref ref-type="fig" rid="fig2s1">Figure 2—figure supplement 1A–B</xref>). Taken together, these results show that when an additional TFBS is added to an existing synthetic element, the position and orientation of the new site can have large effects on activity.</p><fig-group><fig id="fig2" position="float"><label>Figure 2.</label><caption><title>Non-additivity in synthetic elements.</title><p>(<bold>A</bold>) Comparison of synthetic 3-mer elements with matched 4-mer elements containing one additional site in the first or fourth position. Mean expression of elements across barcodes (black dot) is plotted +/- SEM (black whiskers). Green line for comparison to expression of 3-mer; Green transparency highlights SEM of 3-mer shown. Capital letter represents binding site in forward orientation and lower-case letter represents binding site in reverse orientation. Activity of the ten highest (<bold>B</bold>) and ten lowest (<bold>C</bold>) expressing 4-mers. Red line represents average expression of all synthetic 4-mer elements. Case represents binding site orientation as in (<bold>A</bold>) Mean expression of each element across barcodes (black dot) +/- SEM (black whiskers). Activity logos for the top 25% (n = 96) (<bold>D</bold>) and bottom 25% (<bold>E</bold>) of 4-mer synthetic elements. Height of letter is proportional to frequency of site in indicated position. Positions organized from 5’ end (Position 1) to 3’ end (Position 4) of elements.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-41279-fig2-v2.tif"/></fig><fig id="fig2s1" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 1.</label><caption><title>Additional examples of non-additivity in synthetic elements.</title><p>Comparisons of synthetic 3-mer elements with matched 4-mer elements containing one additional site in the first or fourth position with (<bold>A</bold>) three of four matched 4-mers with overlapping expression despite an additional binding site and (<bold>B</bold>) one of four matched 4-mers with overlapping expression. Activity logos for the top 25% (<bold>C</bold>), bottom 25% (<bold>D</bold>) of 3-mer synthetic elements (n = 48 each), and top 25% (<bold>E</bold>) and bottom 25% (<bold>F</bold>) of 2-mer synthetic elements (n = 12 each). Height of letter is proportional to frequency of site in indicated position. Positions organized as in <xref ref-type="fig" rid="fig2">Figure 2</xref>.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-41279-fig2-figsupp1-v2.tif"/></fig></fig-group><p>Synthetic elements appear to follow a grammar that includes some position specific interactions between TFBS. The ten highest expressing elements in the SYN library all have S and O sites next to each other and in the first two positions (<xref ref-type="fig" rid="fig2">Figure 2B</xref>), while the ten lowest expressing 4-mers have a strong bias for O and S in the last two positions (<xref ref-type="fig" rid="fig2">Figure 2C</xref>). The 10 highest expressing 4-mers all have K followed by E in the last two positions, while the lowest expressing 4-mers tend to have K and E in the first two positions. The fourth position can have an especially large effect on expression. In the highest 25% of 4-mers S is depleted (0/96) in the fourth position (<xref ref-type="fig" rid="fig2">Figure 2D</xref>), while in the lowest 25% E is virtually depleted (1/96) in the fourth position (<xref ref-type="fig" rid="fig2">Figure 2E</xref>). Conversely, in the fourth position, E is overrepresented in the top 25% (64/96) while S is overrepresented in the bottom 25% (48/96). These patterns also hold for comparisons of the strongest and weakest 3-mer and 2-mer elements (<xref ref-type="fig" rid="fig2s1">Figure 2—figure supplement 1C–F</xref>). These patterns indicate a grammar that includes a bias for S and O sites positioned upstream of K and E sites. This positioning may favor interactions between these factors and the basal transcriptional machinery or TFs recruited by the minimal promoter. As specifying a site at a given position restricts possible sites in neighboring positions, these patterns could also represent favorable interactions between factors. These data show that the precise arrangement of TFBS influences the activities of synthetic elements.</p></sec><sec id="s2-5"><title>Modeling supports a role for TFBS positions in setting expression level for synthetic elements but not for genomic sequences</title><p>While the grammar of O, S, K, and E sites influences the relative activities of the SYN elements, their order and orientation does not appear to contribute to the activity of genomic sequences. We compared the SYN and gWT libraries for elements with configurations of OKE, OSE, OSK, and SKE TFBS. Unlike SYN 3-mer elements, all four classes of gWT sequences span the full range of expression levels observed for the entire library, with only OSK sequences having a higher average expression (<xref ref-type="fig" rid="fig3s1">Figure 3—figure supplement 1A</xref>). Thus, in genomic sequences, the same arrangement of sites embedded in different genomic contexts can either fail to drive detectable activity or drive expression higher than the highest SYN library member. To quantify the divergence in activities between genomic and synthetic elements directly, we matched gWT sequences with pluripotency TFBS-dependent activity to SYN elements with the corresponding order of TFBS. We observed no correlation in regulatory activity between matched site configurations, (R<sup>2</sup> = 0.001; <xref ref-type="fig" rid="fig3s1">Figure 3—figure supplement 1B</xref>). These data indicate that other variables contribute to the <italic>cis</italic>-regulatory activity of gWT sequences, such as the spacing and affinities of the sites, or the presence of TFBS for additional factors in flanking sequences that are held constant in the SYN library.</p><p>To identify additional sequence features that might be contributing to activity, we used a variation of the Random Forest (RF) model, an unsupervised machine learning technique. RF models can be applied for either simple classification, assigning observations to group predictions, or classifying individual observations into semi-continuous bins to make quantitative, regression-case predictions. The accuracy of predictions are assessed over a large number of decision trees trained on random subsets of the data, which allows the contribution or ‘variable importance’ of specific features to be measured. As RFs are prone to biases from early random splits in the decision trees for unbalanced data, we used iterative Random Forests (iRF) as a tool for feature selection as well as for predicting activity (<xref ref-type="bibr" rid="bib2">Basu et al., 2018</xref>).</p><p>We first trained a regression-case iRF model on the data from the SYN library. We initialized the models with four features (<xref ref-type="supplementary-material" rid="supp2">Supplementary file 2A</xref>), representing only the presence or absence of each of the four pluripotency TFBS. This ‘independent’ iRF model had an R<sup>2</sup> of 0.56 between observed and predicted observations when tested on held-out data for the final iRF iteration (<xref ref-type="fig" rid="fig3s2">Figure 3—figure supplement 2</xref>). However, the independent iRF model cannot account for the differences in activities between 4-mers, because all 4-mers have identical TFBS composition (4-mers R<sup>2</sup> = 0.00). To identify features that might distinguish between the activities of 4-mers, we trained an additional regression-case iRF model, ‘independent + position’, initialized with 20 features, representing both the presence and position of the four TFBS in each SYN element (<xref ref-type="supplementary-material" rid="supp2">Supplementary file 2A</xref>). The 20-term positional model performs well in predicting SYN expression, with an overall R<sup>2</sup> of 0.87 for the last model iteration on a held-out test set (<xref ref-type="fig" rid="fig3">Figure 3A</xref>). The positional iRF model highly weights the presence/absence of the sites, as expected from the performance of the independent iRF model, but also has contributions from the presence of E in the 4th position and S in the first and second positions (<xref ref-type="fig" rid="fig3">Figure 3B</xref>). These results reinforce the conclusion that the activity of synthetic sequences depends both on the composition and positioning of TFBS.</p><fig-group><fig id="fig3" position="float"><label>Figure 3.</label><caption><title>Positional grammar in synthetic elements.</title><p>(<bold>A</bold>) Iterative random forest (iRF) regression model that includes features for presence and position of pluripotency TFBS predicts relative expression of synthetic elements. Number of binding site per element is indicated in pink (2-mers), green (3-mers), and blue (4-mers). Observed and predicted expression are both plotted in log<sub>2</sub> space. (<bold>B</bold>) Ranking of variables in synthetic iRF model. Variable importance is estimated by Increased Node Purity (IncNodePurity), the decrease in node impurities from splitting on that variable, averaged over all trees during training.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-41279-fig3-v2.tif"/></fig><fig id="fig3s1" position="float" specific-use="child-fig"><label>Figure 3—figure supplement 1.</label><caption><title>Comparison of synthetic and genomic patterns of transcription factor binding sites (TFBS).</title><p>(<bold>A</bold>) Expression (log<sub>2</sub>) of all synthetic (dark blue) and gWT (dark green) library members subset by TFBS composition (light blue and light green, respectively). (<bold>B</bold>) Expression (log<sub>2</sub>) of synthetic (x-axis) and gWT (y-axis) library members, matched by composition and order of binding sites for OCT4 (O), SOX2 (S), KLF4 (K), and ESRRB (E). Subsets of TFBS composition indicated by color. Gray line indicates x-y diagonal as axis scales differ.133.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-41279-fig3-figsupp1-v2.tif"/></fig><fig id="fig3s2" position="float" specific-use="child-fig"><label>Figure 3—figure supplement 2.</label><caption><title>Additive effects in synthetic elements.</title><p>Iterative random forest (iRF) regression model that includes features for only presence of pluripotency TFBS to predict the relative expression of held out test set of synthetic elements. Number of binding site per element indicated as in <xref ref-type="fig" rid="fig3">Figure 3</xref>. Observed and predicted expression are both plotted in log<sub>2</sub> space.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-41279-fig3-figsupp2-v2.tif"/></fig><fig id="fig3s3" position="float" specific-use="child-fig"><label>Figure 3—figure supplement 3.</label><caption><title>Effect of spacer sequences between TFBS on synthetic 4-mer expression.</title><p>(<bold>A</bold>) Expression of sequences in ‘mini spacer’ library with different binding sites. (<bold>B</bold>) Difference in expression between each 4-mer oligo with new spacer and the original spacer. (<bold>C</bold>) Expression of each 4-mer oligo with original and new spacers. The numbers next to each point indicate the expression rank of each oligo with its original spacer, with one being the highest expressed and six being the lowest.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-41279-fig3-figsupp3-v2.tif"/></fig></fig-group><p>iRF models trained on the SYN library failed to predict or classify the expression of genomic sequences. While synthetic elements had a range of activities, elements in the gWT library are predominantly inactive, and the small number of active gWT sequences drive expression across an order of magnitude of activity levels (<xref ref-type="fig" rid="fig3s1">Figure 3—figure supplement 1A</xref>). Having such a large number of inactive sequences in the pool makes it difficult to train a model that predicts the relative activities of genomic sequences. Retraining iRF regression models to predict gWT expression fails during the training step and has no correlation with the observed expression data (independent: R<sup>2</sup> = 0.03; independent + position: R<sup>2</sup> = 0.001). In all subsequent analyses of genomic sequences, we limited ourselves to models that attempt to distinguish between active and inactive genomic sequences, without predicting the relative differences in activity among active sequences. However, our first attempt to produce a classifier failed. Training a classification model to distinguish between active and inactive gWT sequences (top 25%, n = 102; bottom 75%, n = 305) using either only independent or independent + position features also fails to perform better than chance (Independent: Area Under the Receiver Operator Curve (AUROC) = 0.52, Area Under the Precision Recall Curve (AUPRC) = 0.22; Positional: AUROC = 0.47, AUPRC = 0.25; <xref ref-type="supplementary-material" rid="supp2">Supplementary file 2B</xref>). Genomic and synthetic elements with the same pattern of sites can drive drastically different expression levels (<xref ref-type="fig" rid="fig3s1">Figure 3—figure supplement 1B</xref>). Other sequence features present in the flanking genomic sequences and absent from the synthetic elements must therefore play a role in setting activity levels, in addition to the identity and position of the individual pluripotency TFBS.</p><p>Our results with genomic elements suggested that the sequences flanking the pluripotency TFBS play a role in determining <italic>cis</italic>-regulatory activity. We tested the effect of changing spacer sequences that flank the TFBS in six 4-mer elements from the SYN library. We tested four different spacer sequences, for a total of 30 library members, which includes the original spacer sequence. The new spacers sequences were designed to match the nucleotide content of the original spacers and minimize the creation of new TFBS (<xref ref-type="supplementary-material" rid="supp1">Supplementary file 1J</xref>). To ensure the dynamic range of the library, we mixed this ‘mini spacer library’ library with a small portion of the SYN library and performed an MPRA.</p><p>We found that changing the spacer sequences in the SYN library had small, but significant effects on the activities of the 4-mers. The activities of all six 4-mers in the mini spacer library tested with all four spacer sequences remained in the original range of expression for 4-mers (<xref ref-type="fig" rid="fig3s3">Figure 3—figure supplement 3A</xref>). On average, the spacer sequences modified expression by 6% (0.3–25%, <xref ref-type="fig" rid="fig3s3">Figure 3—figure supplement 3B</xref>). Although the overall effects of spacer sequences were small, the rank order of the 4-mers did change for different spacers (<xref ref-type="fig" rid="fig3s3">Figure 3—figure supplement 3C</xref>), supporting the idea that sequence features flanking the binding sites do affect gene expression. These results are consistent with the differences between the SYN and gWT libraries.</p></sec><sec id="s2-6"><title>Site affinity contributes to the activity of genomic sequences</title><p>We attempted to identify other sequence features that might differentiate active and inactive gWT sequences. Sequence-based support vector machines (<italic>k</italic>mer-SVMs) are powerful tools to predict the activity of regulatory elements (<xref ref-type="bibr" rid="bib16">Fletez-Brant et al., 2013</xref>; <xref ref-type="bibr" rid="bib4">Chaudhari and Cohen, 2018</xref>). To identify sequence features that explain the differences between genomic elements, we trained a gapped <italic>k</italic>mer SVM (gkm-SVM) (<xref ref-type="bibr" rid="bib19">Ghandi et al., 2016</xref>; <xref ref-type="bibr" rid="bib18">Ghandi et al., 2014</xref>). The best performing gkm-SVM classified our positive and negative sets with AUROC of 0.75 and AUPRC of 0.77 (<italic>k</italic> = 8, gap = 2; <xref ref-type="fig" rid="fig4">Figure 4A</xref>). Although all sequences in the gWT library were selected to contain TFBS for the four pluripotency factors, many of the discriminative 8-mers (29/50) have motif matches that include at least one pluripotency family member (<xref ref-type="bibr" rid="bib16">Fletez-Brant et al., 2013</xref>; <xref ref-type="bibr" rid="bib1">Bailey et al., 2009</xref>; <xref ref-type="supplementary-material" rid="supp2">Supplementary file 2D</xref>). This suggests that the differences between active and inactive genomic sites could be due to the primary pluripotency sites or secondary occurrences of these sites in the intervening sequences that scored below the scanning threshold.</p><fig-group><fig id="fig4" position="float"><label>Figure 4.</label><caption><title>Sequence features separate active and inactive genomic sequences.</title><p>(<bold>A</bold>) Performance of gkm-SVM for genomic sequences supports contribution of sequence-based features to activity. Word length of 8 bp with gap size of 2 bp was used for training with threefold cross validation. ROC curve (left panel) and PR curve (right panel) is plotted for the average across threefold cross-validation sets +/- standard deviation. (<bold>B–E</bold>) Primary (O,S,K,E) site affinities across gWT sequences, as output during motif scanning plotted for high genomic sequences (top 25% as ranked by expression, n = 101) and low genomic sequences (bottom 25% as ranked by expression, n = 101). (<bold>F–G</bold>) Total site affinities is calculated per sequence by summing the predicted affinity of the three primary sites present in each sequence. (<bold>H</bold>) Total number of occurrences of TFBS for additional TFs in high and low sequences (stratified as in <bold>B–G</bold>), as determined by motif scanning, excluding primary (O,S,K,E) sites.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-41279-fig4-v2.tif"/></fig><fig id="fig4s1" position="float" specific-use="child-fig"><label>Figure 4—figure supplement 1.</label><caption><title>Predicted occupancy of genomic sequences.</title><p>Predicted occupancy (P(Occ)) for genomic sequences in the absence of the primary pluripotency sites (gMUT sequences) for high assumed protein concentration (mu) for SOX2 (mu = 8), OCT4 (mu = 10), KLF4 (mu = 8), and ESRRB (mu = 8) shown in middle and right panels. Summed P(Occ) of all factors per gMUT sequence, compared to expression (top left panel) or binned as low or high library members (bottom 25% and top 25% of sequences, ranked by gWT expression, n = 101).</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-41279-fig4-figsupp1-v2.tif"/></fig><fig id="fig4s2" position="float" specific-use="child-fig"><label>Figure 4—figure supplement 2.</label><caption><title>Genomic sequences show distance preferences between factors.</title><p>Comparison of fraction of sequences (density) with designated edge to edge spacing between S, O, K, and E sites. Site positions outputted by scanning high (top 25% as ranked by gWT expression, n = 101) and low (bottom 25% as ranked by gWT expression, n = 101) sequences. Top two panels show fraction of sequences with indicated distances between site positions relative to the promoter, regardless of identity. Bottom six panels show fraction of sequences with indicated distances between adjacent sites, accounting for site identities.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-41279-fig4-figsupp2-v2.tif"/></fig></fig-group><p>Sequences with higher predicted affinity pluripotency TFBS may drive higher expression. To determine if differences in the primary pluripotency sites are part of the signal identified by the SVM, we annotated gWT sequences with PWM-based scores for each TFBS present (<xref ref-type="bibr" rid="bib21">Grant et al., 2011</xref>). For SOX2, we found no difference in scores between high and low sequences (<xref ref-type="fig" rid="fig4">Figure 4B</xref>; p=0.07, Welch’s t-test). For OCT4, we found a modest difference between the average scores for high and low sequences and a broader but also a significant difference for KLF4 and ESRRB PWM scores (<xref ref-type="fig" rid="fig4">Figure 4C–E</xref>). Summing the PWM scores for all of the TFBS further separates high and low sequences (<xref ref-type="fig" rid="fig4">Figure 4F–G</xref>). These patterns suggest that the quality of the primary sites contributes to the activity differences observed among gWT sequences.</p><p>We then asked if secondary sites for the pluripotency TFs might contribute to <italic>cis</italic>-regulatory activity by calculating predicted occupancy for both gWT sequences and gMUT sequences that lack the primary binding sites (Materials and methods). Predicted occupancy is a metric that includes contributions from any primary, well-scoring TFBS plus contributions from weaker sites that might be missed with traditional motif scanning (<xref ref-type="bibr" rid="bib58">White et al., 2016</xref>; <xref ref-type="bibr" rid="bib57">White et al., 2013</xref>; <xref ref-type="bibr" rid="bib12">Evans et al., 2012</xref>; <xref ref-type="bibr" rid="bib47">Segal et al., 2008</xref>; <xref ref-type="bibr" rid="bib63">Zhao et al., 2009</xref>). We found evidence for additional low predicted affinity sites for SOX2 and OCT4 in both high and low sequences, making it unlikely that low-affinity sites strongly contribute to expression differences (<xref ref-type="fig" rid="fig4s1">Figure 4—figure supplement 1</xref>). Together, these results suggest that the affinities of the primary sites in genomic sequences, which are fixed in synthetic elements, contribute to the regulatory activity of genomic sequences more than the presence of additional sites with low predicted affinity.</p><p>We also analyzed whether the spacing between binding sites correlated with the activity of <italic>cis</italic>-regulatory elements. Using the same annotations used to determine the predicted affinities of SOX2, OCT4, ESRRB, and KLF4 binding sites, we calculated the edge-to-edge distance between every possible pair of binding sites and plotted the frequency of each spacing for high and low activity sequences (<xref ref-type="fig" rid="fig4s2">Figure 4—figure supplement 2</xref>). We observed a preference in high activity sequences for closely spaced sites for OCT4 and SOX2 reflecting a known interaction between these TFs. We also observed preferences in high activity genomic sequences for closely spaced KLF4 and OCT4 sites, and for ESRRB and OCT4 sites. Binding site spacing may therefore play a role in setting the relative activities of genomic sequences.</p></sec><sec id="s2-7"><title>Contributions from sites for other transcription factors</title><p>A major difference between the synthetic and genomic elements is the presence of sites for TFs besides the pluripotency factors. While the synthetic elements were designed to keep the sequences between pluripotency sites constant, genomic sequences differ in both the length and composition of sequences between the pluripotency sites. The presence of binding sites for additional transcription factors may contribute to the activity of genomic sequences. To identify sites for other factors that could contribute to differences between high and low activity gWT sequences, we examined the top discriminative 8-mers from the gkm-SVM, looking at possible PWM matches for additional TFs (<xref ref-type="supplementary-material" rid="supp2">Supplementary file 2D</xref>). We then used PWMs for these additional TFs to identify instances of sites for other factors in the genomic sequences (see Materials and methods) (<xref ref-type="bibr" rid="bib21">Grant et al., 2011</xref>; <xref ref-type="bibr" rid="bib46">Sandelin, 2004</xref>). We found significant enrichment for FOXA1 sites (<xref ref-type="fig" rid="fig4">Figure 4H</xref>). We also found that FOXA1 and NANOG had higher total PWM scores in the high activity sequences (<xref ref-type="fig" rid="fig5s1">Figure 5—figure supplement 1A</xref>). While FOXA1 is likely not present in mESCs, other family members (FOXA2, FOXD1, FOXP1) are expressed in ESCs and have been shown to contribute to the pluripotent regulatory network, and therefore could be acting on the gWT sequences through these binding sites (<xref ref-type="bibr" rid="bib41">Pan and Thomson, 2007</xref>; <xref ref-type="bibr" rid="bib39">Mulas et al., 2018</xref>; <xref ref-type="bibr" rid="bib17">Gabut et al., 2011</xref>).</p><p>Genomic sequences with higher occupancy by TFs in the genome, as measured by ChIP-seq, have higher average expression in our assay. We annotated the gWT intervals with publicly available ChIP-seq data for additional TFs and with ATAC-seq data from E14 mESCs to determine if differences in accessibility explained the difference between high and low activity sequences (<xref ref-type="supplementary-material" rid="supp2">Supplementary file 2B</xref>). Both high and low activity gWT sequences were accessible in the genome showing that accessibility does not necessarily correlate with high activity sequences. High activity sequences had a small but significant overlap with NANOG peaks (<xref ref-type="fig" rid="fig5s1">Figure 5—figure supplement 1B</xref>). However, for the 328 genomic sequences with a NANOG ChIP-seq signal, only 16% had an underlying TFBS as determined by motif scanning. Therefore, NANOG might be recruited by other pluripotency TFs to these sequences independent of high-quality TFBS for this factor. If we compare expression levels to the number of overlapping ChIP-seq peaks, including O,S,K,E and these additional TFs, we see that gWT sequences with higher occupancy in the genome have higher average expression in our assay (<xref ref-type="fig" rid="fig5">Figure 5</xref>), which has been previously observed in HepG2 cells (<xref ref-type="bibr" rid="bib52">Ulirsch et al., 2016</xref>). This result supports a model where cumulative occupancy sets activity level.</p><fig-group><fig id="fig5" position="float"><label>Figure 5.</label><caption><title>Activity of genomic sequences scales with increased occupancy in the genome.</title><p>Expression of elements binned by number of intersected ChIP-seq peak signals for different factors. Number of sequences in each bin indicated in center of boxplot. All gWT sequences overlapped at least one ChIP-seq peak as per library design.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-41279-fig5-v2.tif"/></fig><fig id="fig5s1" position="float" specific-use="child-fig"><label>Figure 5—figure supplement 1.</label><caption><title>Genomic sequences show signatures for other factors.</title><p>(<bold>A</bold>) Summed motif scores for indicated motif across genomic sequences, excluding primary pluripotency sites. Site scores output during motif scanning of high (top 25% as ranked by gWT expression, n = 101) and low (bottom 25% as ranked by gWT expression, n = 101) gMUT sequences to prevent scoring of O, S, K, or E TFBS sequences. (<bold>B</bold>) Overlapping TF occupancy, as measured by ChIP-seq, or accessibility, as measured by ATAC-seq, for high (top 25% as ranked by gWT expression, n = 101) and low (bottom 25% as ranked by gWT expression, n = 101) genomic sequence intervals.</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-41279-fig5-figsupp1-v2.tif"/></fig></fig-group><p>To understand the relative contributions of the sequence features that were enriched individually, we trained iRF models with different subsets of these sequence features and compared their performance on a held-out test set (<xref ref-type="supplementary-material" rid="supp2">Supplementary file 2B</xref>). None of these models accurately predicted the activity of genomic sequences, likely because most genomic sequences in our collection had no activity above basal levels. Therefore, we attempted to classify active from inactive genomic sequences.</p><p>We trained an iRF model initialized with 58 features that capture differences between gWT sequences and SYN elements. These features include predicted affinity and preferred spacings between the pluripotency TFBS, the predicted occupancy for the pluripotency TFs, the presence of binding sites for additional TFs, plus chromatin accessibility (ATAC-seq) and ChIP-seq peaks for both TFs and histone marks, as well as summary features such as the total primary site affinities for each sequence (<xref ref-type="supplementary-material" rid="supp2">Supplementary file 2B</xref>). This gWT iRF model classified active from inactive on a held out test set with AUROC = 0.67, and AUPRC = 0.46 (<xref ref-type="fig" rid="fig6">Figure 6A–B</xref>, model ‘All’). Models that only included subsets of features — the spacing between elements (model ‘Spacing’), the strength of the pluripotency sites (‘PrimarySites’), or the overlapping ChIP signal (‘ChIPSignals’) — did not perform as well (<xref ref-type="fig" rid="fig6">Figure 6A–B</xref>). The features that best separate active from inactive sequences were related to attributes of the pluripotency sites with the top feature being the summed pluripotency factor predicted affinity per sequence (‘OSKE_TotalAffinity’, <xref ref-type="fig" rid="fig6">Figure 6C</xref>). Taken together, our data suggest that genomic sequences drive higher expression when they contain strong binding sites with preferred spacing and are embedded in sequences that can mediate the recruitment of other TFs or cofactors.</p><fig id="fig6" position="float"><label>Figure 6.</label><caption><title>Performance of iRF classification models that include features specific to genomic sequences.</title><p>(<bold>A</bold>) ROC Curve and (<bold>B</bold>) Precision-Recall (PR) Curve comparing genomic iRF models. Color indicates set of features used to train model. (<bold>C</bold>) Variable importance as evaluated for the feature by the average reduction in the Gini index (<xref ref-type="bibr" rid="bib7">Chen et al., 2008c</xref>).</p></caption><graphic mime-subtype="tiff" mimetype="image" xlink:href="elife-41279-fig6-v2.tif"/></fig></sec></sec><sec id="s3" sec-type="discussion"><title>Discussion</title><p>In this study, we sought to understand how pluripotency factors collaborate to drive specific levels of expression by testing both an exhaustive set of synthetic arrangements of TFBS for OCT4, SOX2, KLF4, and ESRRB and comparable genomic sequences. The experimental design allowed for direct comparisons between the regulatory grammar of synthetic and genomic sequences. The strongest similarity between synthetic and genomic elements is that in both cases activity depends heavily on the number and affinity of binding sites. These results are most consistent with a model in which the overall occupancy of a sequence by its cognate TFs is the primary determinant of that element’s activity. Consistent with this hypothesis, the predictive power of our trained genomic model derived primarily from summing over the number and affinity of binding sites. We also observed correlation between the occupancy of sites as measured by ChIP-seq and their activity in MPRA assays. While there are many steps involved in activating gene expression, the occupancy model posits that the strength of a regulatory element is primarily controlled by its fractional occupancy by TFs.</p><p>The occupancy model might also explain the surprising result that the activity of genomic elements in our plasmid MPRA experiments do not correlate with experimental measurements of how accessible the chromatin is in their native locations. Plasmid assays might not capture regulation by chromatin, but in many cases plasmid assays do recapitulate the activity of chromosomally integrated elements (<xref ref-type="bibr" rid="bib36">Maricque et al., 2019</xref>; <xref ref-type="bibr" rid="bib26">Inoue et al., 2017</xref>). Alternatively, accessible regions may be bound by transcription factors but may not necessarily drive activity, such as in the case of ‘poised’ regulatory elements (<xref ref-type="bibr" rid="bib9">Cruz-Molina et al., 2017</xref>). Nucleosome exclusion is important for regulatory activity (<xref ref-type="bibr" rid="bib29">Khoueiry et al., 2010</xref>) and may reflect TF binding, but accessibility itself may not be sufficient for regulatory activity. Another possibility is that open chromatin may not be a direct reflection of the occupancy of an element by its cognate TFs. Other factors besides occupancy by TFs also determine the openness of chromatin, such as chromosome topology, the proximity of origins of replication, and nucleotide composition. This may explain why some genomic sequences with binding sites that reside in open chromatin do not drive high activity in MPRA assays. The prediction is that these regions are open for reasons other than occupancy by cognate TFs. That the activity of genomic elements correlates with TF occupancy as measured by ChIP-seq, but not necessarily open chromatin measurements by ATAC-seq, supports the occupancy model.</p><p>While TF occupancy was the best predictor of activity, the AUROC and AUPRC analyses show that we are still missing important features that underlie the activity of genomic sequences. Indeed, two-thirds of genomic sequences that contain consensus motifs and reside under a ChIP-seq peak for one of the pluripotency TFs had no activity in our assay. Why don’t all sequences occupied by TFs have strong regulatory activity? The sequence context in which occupied binding sites occur must contribute heavily to their activity. We attempted to address this issue by examining the regulatory grammar of synthetic elements.</p><p>Synthetic elements provide a highly controlled system for exploring whether TFBS are constrained by a regulatory grammar. With synthetic elements we found clear evidence that their activity depends on the position and orientation of pluripotency binding sites. Synthetic elements with the same number and affinity of TFBS had different levels of activity depending on the order and orientation of the sites. This result suggests that active regulatory elements in the genome are defined not only by the presence of TF occupied motifs, but also by cues in the surrounding DNA sequences. However, our models that captured the specific regulatory grammar of synthetic elements failed to predict the activity of genomic sequences.</p><p>Why don’t models that robustly predict the activity of synthetic elements also predict the activity of genomic sequences? With synthetic elements, each sequence differs from others in the library by only a small number of sequence features. In synthetic libraries, there are many pairs of elements that differ by only a single sequence feature, which provides power to observe experimentally the effect of a single variable. In contrast, libraries of genomic elements are much more diverse, and the analysis of genomic sequences relies on detecting correlations between elements that share sequence features. However, it is difficult to isolate the effect of a single sequence feature because genomic elements that share a certain sequence feature will always be very different in terms of other features. The strength of the synthetic approach is the power it provides to isolate the effects of specific sequence features or pairs of sequence features. The weakness of the synthetic approach is that genomic elements are subject to many context specific constraints, all of which cannot be captured in a single synthetic library. When we changed the spacer sequences in our synthetic library, we found small but reproducible effects on expression. Our interpretation of this result is that changing the spacer sequences did not have large effects on the independent contribution of each TFBS, but did have effects on the interactions between sites (i.e. the regulatory grammar). In the future, we plan to use the regulatory grammar derived from synthetic elements to design experiments that manipulate single features of genomic elements. If the grammar that is learned from synthetic elements reflects real constraints in the cell, then models of synthetic elements should predict the relative effects of single perturbations of genomic elements even if they cannot predict the absolute expression of genomic sequences. A combined approach that leverages both synthetic and genomic sequences should continue to help unravel the rules that govern <italic>cis</italic>-regulation of expression in cells.</p></sec><sec id="s4" sec-type="materials|methods"><title>Materials and methods</title><table-wrap id="keyresource" position="anchor"><label>Key resources table</label><table frame="hsides" rules="groups"><thead><tr><th valign="top">Reagent type <break/>(species) or resource</th><th valign="top">Designation</th><th valign="top">Source or reference</th><th valign="top">Identifiers</th><th valign="top">Additional information</th></tr></thead><tbody><tr><td valign="top">Cell line (<italic>Mus musculus</italic> mouse)</td><td valign="top">RW4</td><td valign="top">other</td><td valign="top">RRID:<ext-link ext-link-type="uri" xlink:href="https://scicrunch.org/resolver/CVCL_6442">CVCL_6442</ext-link></td><td valign="top">Gift from Mitra Lab, CGS, Department of Genetics, Washington University School of Medicine in St. Louis. The cell line tested <break/>negative for mycoplasma <break/>contamination by the Genome Editing <break/>and iPSC core at Washington University <break/>in St. Louis.</td></tr><tr><td valign="top">Commercial assay or kit</td><td valign="top">PureLink RNA Mini Kit</td><td valign="top">ThermoFisher Scientific/Invitrogen</td><td valign="top">Cat#:12183018A</td><td valign="top">Followed manufacturer’s protocol</td></tr><tr><td valign="top">Commercial assay or kit</td><td valign="top">PureLink DNase Set</td><td valign="top">ThermoFisher Scientific/Invitrogen</td><td valign="top">Cat#:12185010</td><td valign="top">Followed manufacturer’s protocol</td></tr><tr><td valign="top">Commercial assay or kit</td><td valign="top">TURBO DNA-free</td><td valign="top">ThermoFisher Scientific/Invitrogen</td><td valign="top">Cat#:AM1907</td><td valign="top">Followed manufacturer’s protocol</td></tr><tr><td valign="top">Commercial assay or kit</td><td valign="top">SuperScript III Reverse Transcriptase</td><td valign="top">ThermoFisher Scientific/Invitrogen</td><td valign="top">Cat#:18080044</td><td valign="top">Followed manufacturer’s protocol</td></tr><tr><td valign="top">Commercial assay or kit</td><td valign="top">anti-Alkaline Phosphatase (AP) staining</td><td valign="top">System Biosciences</td><td valign="top">Cat.#:AP100R-1</td><td valign="top">Followed manufacturer’s protocol</td></tr><tr><td valign="top">Recombinant DNA reagent</td><td valign="top">SYN</td><td valign="top">this paper</td><td valign="top"/><td valign="top">Recombinant plasmid library of synthetic (SYN) elements upstream of a minimal Pou5f1 promoter and dsRed/SV40 UTR reporter element</td></tr><tr><td valign="top">Recombinant DNA reagent</td><td valign="top">GEN</td><td valign="top">this paper</td><td valign="top"/><td valign="top">Recombinant plasmid library of sequences identified in the mouse genome (GEN) upstream of a minimal Pou5f1 promoter and dsRed/SV40 UTR reporter element</td></tr><tr><td valign="top">Recombinant DNA reagent</td><td valign="top">miniSpacer</td><td valign="top">this paper</td><td valign="top"/><td valign="top">Recombinant plasmid library of synthetic elements with swapped spacer (miniSpacer) sequences upstream of a minimal Pou5f1 promoter and dsRed/SV40 UTR reporter element</td></tr><tr><td valign="top">Software, algorithm</td><td valign="top">Bedtools v2.2</td><td valign="top"><ext-link ext-link-type="uri" xlink:href="https://bedtools.readthedocs.io/en/latest/">https://bedtools.readthedocs.io/en/latest/</ext-link></td><td valign="top">RRID:<ext-link ext-link-type="uri" xlink:href="https://scicrunch.org/resolver/SCR_006646">SCR_006646</ext-link></td><td valign="top">DOI:<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1093/bioinformatics/btq033">10.1093/bioinformatics/btq033</ext-link></td></tr><tr><td valign="top">Software, algorithm</td><td valign="top">iRF v2.0.0</td><td valign="top"><ext-link ext-link-type="uri" xlink:href="https://cran.r-project.org/web/packages/iRF/index.html">https://cran.r-project.org/web/packages/iRF/index.html</ext-link></td><td valign="top">DOI:<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1073/pnas.1711236115">10.1073/pnas.1711236115</ext-link></td><td valign="top"/></tr><tr><td valign="top">Software, algorithm</td><td valign="top">gkm-SVM</td><td valign="top"><ext-link ext-link-type="uri" xlink:href="https://cran.r-project.org/web/packages/gkmSVM/index.html">https://cran.r-project.org/web/packages/gkmSVM/index.html</ext-link></td><td valign="top">DOI:<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1093/bioinformatics/btw203">10.1093/bioinformatics/btw203</ext-link></td><td valign="top"/></tr><tr><td valign="top">Software, algorithm</td><td valign="top">BEEML</td><td valign="top"><ext-link ext-link-type="uri" xlink:href="http://stormo.wustl.edu/beeml/">http://stormo.wustl.edu/beeml/</ext-link></td><td valign="top">DOI:<ext-link ext-link-type="uri" xlink:href="https://doi.org/10.1371/journal.pcbi.1000590">10.1371/journal.pcbi.1000590</ext-link></td><td valign="top"/></tr></tbody></table></table-wrap><sec id="s4-1"><title>Library design</title><p>To generate a library that contained both synthetic and genomic elements, we ordered a custom pool of 13,000 unique 150 bp oligonucleotides (oligos) from Agilent Technologies (Santa Clara, CA) through a limited licensing agreement. Each oligo in the SYN pool was 150 bp in length with the following sequence:</p><list list-type="simple"><list-item><p><named-content content-type="sequence">CTTCTACTACTAGGGCCCA[SEQ]AAGCTT[FILL]GAATTCTCTAGAC[BC]TGAGCTCTACATGCTAGTTCATG</named-content></p></list-item></list><p>where [SEQ] is a 40–80 bp synthetic element comprised of concatenated 20 bp building blocks of pluripotency sites, as described previously, with the fifth position of the KLF4 site changed to ‘T’ to facilitate cloning (<xref ref-type="bibr" rid="bib14">Fiore and Cohen, 2016</xref>). [FILL] is a random filler sequence of variable length to bring the total length of each sequence to 150 bp, and [BC] is a random 9 bp barcode. The oligonucleotide pool contained all possible combinations of the pluripotency binding sites in both orientations, with no more than one of each site per sequence in lengths of two, three, and four building blocks. The sequence of each of the element is listed in <xref ref-type="supplementary-material" rid="supp1">Supplementary file 1B</xref>. In total, the SYN library has 624 unique synthetic elements. Each synthetic element is present in the pool eight times, each time with a different unique BC. There are also 112 oligos in the pool for cloning the basal promoter without any upstream element, each with a unique BC.</p><p>Genomic sequences were represented in the pool by 150 bp oligos with the following sequences:</p><list list-type="simple"><list-item><p><named-content content-type="sequence">GACTTACATTAGGGCCCGT[SEQ]AAGCTT[FILL]GAATTCTCTAGAC[BC]TGAGCTCGGACTACGATACTG</named-content></p></list-item></list><p>where [SEQ] is either a reference (gWT) or mutated (gMUT) genomic sequence of 81–82 bps. Reference gWT sequences were selected by choosing regions of the genome within 100 bps of previously identified ChIP-seq peaks for these four pluripotency factors (<xref ref-type="bibr" rid="bib6">Chen et al., 2008b</xref>). After excluding poorly sequenced and repetitive regions (<xref ref-type="bibr" rid="bib11">ENCODE Project Consortium, 2012</xref>; <xref ref-type="bibr" rid="bib56">Waterston et al., 2002</xref>), we scanned the remaining regions using FIMO with the four PWMs used previously to design the synthetic building blocks, with a p-value threshold of 1 × 10<sup>−3</sup> (<xref ref-type="bibr" rid="bib21">Grant et al., 2011</xref>; <xref ref-type="bibr" rid="bib1">Bailey et al., 2009</xref>; <xref ref-type="bibr" rid="bib14">Fiore and Cohen, 2016</xref>). Regions that contained more than one overlapping site identified by FIMO were excluded. Binding sites that were located less than 20 bp from each other were then merged into a single genomic element using Bedtools (<xref ref-type="bibr" rid="bib43">Quinlan and Hall, 2010</xref>). Elements with no more than one of each site per element were then selected and expanded to 81–82 bp centered on the motifs. Expanded sequences were rescanned to confirm the presence of only three binding sites with the same threshold as used to originally scan the sequences. Sequences that contained restriction sites for were then removed from the library, leaving 407 genomic sequences with combinations of the OCT4, SOX2, KLF4, and/or ESSRB TFBS.</p><p>We generated matched mutated sequences (gMUT) for each of the 407 gWT sequences by changing two positions in each motif from the highest information content base to the lowest information base for that position (<xref ref-type="fig" rid="fig1s1">Figure 1—figure supplement 1</xref>). The reverse complement position and substitution was made for the reverse orientation of each motif. The mutated sequences were rescanned with all four original PWMs to confirm that no detectable pluripotency TFBS remained, using FIMO with the same p-value threshold (1 × 10<sup>−3</sup>) as above.</p><p>In total, the pool of oligos representing genomic sequences contained 407 wild-type sequences (gWT) and the corresponding 407 gMUT sequences. The sequence of each element is listed in <xref ref-type="supplementary-material" rid="supp1">Supplementary file 1G</xref>. Each of these 814 sequences were associated with eight unique BCs. The primers for gWT and gMUT sequences were identical so all subsequent steps for this library was performed in a single pool. There are also 112 oligos in the pool for cloning the basal promoter without any upstream element, each with a unique BC (<xref ref-type="supplementary-material" rid="supp1">Supplementary file 1F</xref>). The rest of the array contained sequences not used in this study.</p></sec><sec id="s4-2"><title>Cloning of plasmid libraries</title><p>For a full list of primers, see <xref ref-type="supplementary-material" rid="supp3">Supplementary file 3</xref>. The synthesized oligos were prepared as previously described (<xref ref-type="bibr" rid="bib32">Kwasnieski et al., 2012</xref>; <xref ref-type="bibr" rid="bib14">Fiore and Cohen, 2016</xref>), except using primers Synthetic_FW-1 and Synthetic_Rev-2 with an annealing temperature of 55°C for the SYN library and primers Genomic_FW-1 and Genomic_Rev-1 with an annealing temperature of 53°C for the gWT/gMUT libraries. PCR products were purified from a polyacrylamide gel as described previously (<xref ref-type="bibr" rid="bib57">White et al., 2013</xref>). Each library was cloned as described previously (<xref ref-type="bibr" rid="bib14">Fiore and Cohen, 2016</xref>), with an SYN element (SYN library) or either a gWT or gMUT sequence (gWT/gMUT library) cloned into the ApaI and SacI sites of plasmid pCF10.</p><p>The <italic>pou5f1</italic> basal promoter and dsRed reporter gene were amplified from pCF10 using primers CF121 and CF122, and inserted into the plasmid library pools from the previous step at the XbaI and HindIII sites. Digestion of the libraries with SpeI and subsequent size selection was omitted as the SYN library had less than 2% background and the combined gWT/gMUT library had less than 1% background in the final cloning step.</p></sec><sec id="s4-3"><title>Spacer library</title><p>For the mini spacer library, we ordered an oligo pool containing 4-mer elements with different spacer sequences from Integrated DNA Technologies (Coralville, IA). Each oligo in the mini library was 161 bp in length with the following sequence:</p><list list-type="simple"><list-item><p><named-content content-type="sequence">GACATCAAGATCTGGCCTCGGGGCCC[SEQ]AAGCTTGAATTCTCTAGAC[BC]TGAGCTCTCGCTTCGAGCAGACATGAT</named-content></p></list-item></list><p>where [SEQ] represents an oligo sequence described below and [BC] is a random 9 bp barcode. We picked six 4-mer oligos from the original synthetic library to span the 4-mer expression range and swapped out the spacer sequences in the oligos for four other sequences, generating a total of 30 constructs, including the original spacers. Each construct was represented in the pool with five unique barcodes. The sequence of each element is in <xref ref-type="supplementary-material" rid="supp1">Supplementary file 1K</xref>.</p><p>The mini spacer library was cloned into the same backbone as the previous libraries. Briefly, pCF10 was digested with ApaI and SacI, and the single-stranded oligo pool was directly assembled into the backbone using HiFi DNA assembly The <italic>pou5f1</italic> basal promoter and dsRed reporter gene were amplified from pCF10 using CF121 and CF122, then ligated into the mini spacer library following the same approach as the SYN, gWT, and gMUT libraries.</p></sec><sec id="s4-4"><title>Cell culture and transfection</title><p>RW4 mESCs were cultured as described previously (<xref ref-type="bibr" rid="bib60">Xian et al., 2005</xref>; <xref ref-type="bibr" rid="bib5">Chen et al., 2008a</xref>) on 2% gelatin coated plates in standard media (DMEM, 10% fetal bovine serum, 10% newborn calf serum, nucleoside supplement, 1000 U/ml leukemia inhibitory factor (LIF), and 0.1 µM B-mercaptoethanol). Approximately 1 million cells at 100% estimated viability were seeded into six-well plates 24 hr prior to transfection. The SYN library and combined gWT/gMUT were transfected in parallel using 10 µL Lipofectamine 2000 (Life Technologies, Carlsbad, CA), 3 µg of plasmid library, and 0.3 µg CF128 (a GFP control plasmid) per well, as described previously (<xref ref-type="bibr" rid="bib14">Fiore and Cohen, 2016</xref>). Four biological replicates of each library pool, the SYN plasmid pool or combined gWT/gMUT plasmid pool, were transfected and the plates were passaged 6 hr post-transfection. For three replicates of each library pool, RNA was extracted 24 hr post-transfection from approximately 9 million cells per replicate, using the PureLink RNA mini kit (Life Technologies, Carlsbad, CA) with the fourth transfection replicate reserved for estimating transfection efficiency via fluorescent microscopy and staining for alkaline phosphatase (AP) activity, a universal pluripotency marker (<xref ref-type="bibr" rid="bib48">Singh et al., 2012</xref>).</p></sec><sec id="s4-5"><title>Massively parallel reporter assay</title><p>Massively parallel reporter gene assays were used to measure the activity of each element as described previously (<xref ref-type="bibr" rid="bib14">Fiore and Cohen, 2016</xref>; <xref ref-type="bibr" rid="bib38">Mogno et al., 2013</xref>). Briefly, we used Illumina NextSeq (San Deigo, CA) sequencing of both the RNA and original plasmid DNA pool, removing excess DNA from the RNA pool using TURBO DNA-free kit (Life Technologies, Carlsbad, CA). cDNA was then prepared using SuperScript RT III (Life Technologies, Carlsbad, CA) with oligo dT primers. Both the cDNA and the plasmid DNA pool were amplified using primers CF150 and CF151b, for 13 cycles. The PCR amplification products were digested using XbaI and XhoI (New England Biolabs, Ipswich, MA), ligating the resulting digestion products to custom Illumina adapter sequences, P1_XbaI_X (where X is 1 through 8, with in-line multiplexing BC sequences) to the 5’ overhang and PE2_SIC69_SalI on the 3’ XhoI overhang, each of which is comprised of annealed forward (F) and reverse (R) strands. An enrichment PCR with primers CF52 and CF53 was then used, and the resulting products were mixed at equal concentration and sequenced on one NextSeq lane.</p><p>Sequencing reads were filtered to ensure that the BC sequence perfectly matched the expected sequence. For the SYN library, this resulted in 40 million reads combined for the three demultiplexed RNA samples (P1_XbaI_1, P1_XbaI_2, P1_XbaI_3; 12.7–13.5 million each), and 19.7 million reads for the DNA library sample (P1_XbaI_7). For the combined gWT/gMUT libraries, this resulted in approximately 37 million reads combined for the three demultiplexed RNA samples (P1_XbaI_4, P1_XbaI_5, P1_XbaI_6; 9.4–16 million each), and 19.6 million reads for the DNA library sample (P1_XbaI_8). For each library, BCs that had less than three raw counts in any RNA replicate or less than 10 raw counts in the DNA sample were removed before proceeding with downstream analyses.</p><p>Expression normalization was performed by first calculating reads per million (RPM) per BC for each replicate for both the SYN library and the combined gWT/gMUT library. For each BC, expression was calculated by dividing the RPMs in each RNA replicate by the DNA pool RPMs for that BC. Normalizing by DNA RPMs successfully removed the impact of the representation of the construct in the original pool as the calculated expression has no correlation with the DNA counts for both the SYN library and the combined gWT/gMUT. Within each biological replicate, the BCs corresponding to each synthetic element (SYN) or genomic sequence (gWT/gMUT) were averaged and then normalized by basal mean expression in that replicate. These normalized expression values were then averaged across biological replicates. All downstream analyses were performed in R version 3.3.3 and plotted with ggplot2 version 2.2.1. Expression summaries per replicate are reported in <xref ref-type="supplementary-material" rid="supp1">Supplementary file 1C</xref> for the SYN library, <xref ref-type="supplementary-material" rid="supp1">Supplementary file 1H</xref> for the gWT/gMUT library and <xref ref-type="supplementary-material" rid="supp1">Supplementary file 1L</xref> for the ‘mini spacer’ library.</p></sec><sec id="s4-6"><title>Predicted occupancy</title><p>Custom code, based on Zhao and Stormo’s BEEML algorithm (<xref ref-type="bibr" rid="bib63">Zhao et al., 2009</xref>), was used to compare sequences of interest to a provided Energy Weight Matrix (EWM) at a set protein concentration (mu) and output a predicted occupancy for that TF as in <xref ref-type="bibr" rid="bib57">White et al. (2013)</xref>. Briefly, an energy landscape (EWM score) is calculated by comparing all <italic>n</italic>-mers of each sequence, where <italic>n</italic> = length of provided motif, to the matrix to generate an array of individual base scores for the forward and reverse orientation of the sequence. Occupancy is then predicted using equation 3 for binding probability at equilibrium, (<inline-formula><mml:math id="inf1"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mn>1</mml:mn><mml:mrow><mml:mo>/</mml:mo></mml:mrow><mml:mtext> </mml:mtext><mml:mo stretchy="false">(</mml:mo><mml:mn>1</mml:mn><mml:mtext> </mml:mtext><mml:mo>+</mml:mo><mml:mtext> </mml:mtext><mml:msup><mml:mi>e</mml:mi><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mi mathvariant="normal">Δ</mml:mi><mml:mi>G</mml:mi><mml:mtext> </mml:mtext><mml:mo>−</mml:mo><mml:mtext> </mml:mtext><mml:mi>μ</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:msup><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mstyle></mml:math></inline-formula>). Position Frequency Matrices equivalent to the PWMs used for both SYN building block design and for scanning the mouse genome were used to generate EWMs, using the formula <inline-formula><mml:math id="inf2"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mi>R</mml:mi><mml:mi>T</mml:mi><mml:mo>∗</mml:mo><mml:mtext> </mml:mtext><mml:mi>l</mml:mi><mml:mi>n</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mtext> </mml:mtext><mml:mi>F</mml:mi><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:mi>q</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>B</mml:mi><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>e</mml:mi><mml:msup><mml:mspace width="thinmathspace"/><mml:mrow><mml:mrow><mml:mover><mml:mrow/><mml:mo stretchy="false">^</mml:mo></mml:mover></mml:mrow></mml:mrow></mml:msup><mml:mspace width="thinmathspace"/><mml:mi>c</mml:mi><mml:mi>o</mml:mi><mml:mi>n</mml:mi><mml:mi>s</mml:mi><mml:mi>e</mml:mi><mml:mi>n</mml:mi><mml:mi>s</mml:mi><mml:mi>u</mml:mi><mml:mi>s</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mrow><mml:mo>/</mml:mo></mml:mrow><mml:mi>F</mml:mi><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:mi>q</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>B</mml:mi><mml:mi>a</mml:mi><mml:mi>s</mml:mi><mml:mi>e</mml:mi><mml:msup><mml:mspace width="thinmathspace"/><mml:mrow><mml:mrow><mml:mover><mml:mrow/><mml:mo stretchy="false">^</mml:mo></mml:mover></mml:mrow></mml:mrow></mml:msup><mml:mspace width="thinmathspace"/><mml:mi>i</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:mstyle></mml:math></inline-formula> to convert the frequency of each base at each position <italic>i</italic> to a pseudo ΔΔG values for each factor (<xref ref-type="bibr" rid="bib57">White et al., 2013</xref>). Predicted occupancy (P(Occ)) for the 3-mer SYN elements was calculated for different assumed protein concentrations (<italic>mu</italic> = 0.5, 1, 2, 4, 5, 8, 10, 12) to determine at what point the SYN elements are predicted to be saturated, where P(Occ) ≅ three for each SYN element, that is: approaching one for each TFBS in the sequence. SYN elements were saturated by each of the four pluripotency factors at <italic>mu</italic> = 8 with the exception of the shorter Oct4 motif, which reached saturation at <italic>mu</italic> = 10. Occupancy of gWT and gMUT sequences was predicted for gWT and gMUT at an assumed high protein concentration of <italic>mu</italic> = 8 for Sox2, Klf4, Esrrb, and <italic>mu</italic> = 10 for Oct4, consistent with the role of these factors in mESCs. The predicted occupancy of each factor for matched gMUT sequences are reported in <xref ref-type="supplementary-material" rid="supp2">Supplementary file 2F</xref> as a feature of gWT sequences. iRF models:</p><p>We built iterative Random Forest (iRF) models to classify our data using the R package iRF (version 2.0.0) (<xref ref-type="bibr" rid="bib2">Basu et al., 2018</xref>). To run the software a model is initialized with 1/<italic>p</italic> weights for each of <italic>p</italic> features to be included in fitting the model. In each iteration, <italic>p</italic> features are reweighted by their Gini Importance (<italic>w<sup>k</sup></italic>), a measure that is calculated by how purely a node, split by feature, separates the classes (<xref ref-type="bibr" rid="bib37">Menze et al., 2009</xref>; <xref ref-type="bibr" rid="bib34">Louppe et al., 2013</xref>). Default settings were used for model training, with four iterations of reweighting <italic>p</italic> features specified for each model as indicated in <xref ref-type="supplementary-material" rid="supp2">Supplementary files 2A and 2B</xref>.</p><p>Synthetic data was split into training and test sets by randomly subsetting 50% of the total SYN elements (total n = 407). Mean normalized expression was the response variable for model fitting for the synthetic models (see <xref ref-type="supplementary-material" rid="supp2">Supplementary file 2E</xref> for feature annotations for SYN elements). Four iterations of model fitting on training data was used.</p><p>Genomic data was split into training and test sets by randomly subsetting 50% of the total gWT/gMUT intervals (total n = 624). Classification as ‘active’, 1, if mean normalized gWT expression was greater than or equal to the 3rd quartile and ‘inactive’, 0, if mean normalized gWT expression was less than the 3rd quartile (cutoff value = 1.983), was the response variable for model fitting (see <xref ref-type="supplementary-material" rid="supp2">Supplementary file 2F</xref> for feature annotations and response values for gWT sequences). Four iterations of model fitting on training data was used. gkm-SVM:</p><p>We used a gapped <italic>k</italic>-mer Support Vector Machine (gkm-SVM) to search for gapped k-mers that distinguish between highly active and inactive genomic sequences (<xref ref-type="bibr" rid="bib19">Ghandi et al., 2016</xref>). We subset sequences from the gWT library into top 25% (high) and bottom 25% (low) based on expression data for a total of 101 positive and 101 negative intervals for the training set. FASTA sequences were then generated from the mm10 reference genome (Bioconductor, BioMart) for each region (<xref ref-type="supplementary-material" rid="supp4">Supplementary file 4</xref>). We then used the gkm-SVM R package to classify high vs. low sequences (<xref ref-type="bibr" rid="bib19">Ghandi et al., 2016</xref>). Word length (L) values of 6 (gap = 2), 8 (gap = 2), and 12 (gap = 6), were tested with cross validation. Default settings were used for other function options. Three-fold cross validation was chosen due to the the amount of structure in the data, with combinations of OSK binding sites overrepresented in positive training sequences (<xref ref-type="fig" rid="fig3s1">Figure 3—figure supplement 1</xref>). The best average performance on training data as evaluated by AUCs was the model trained with parameters of L = 8 and gap = 2 (See <xref ref-type="supplementary-material" rid="supp2">Supplementary file 2G</xref> for output scores). The final gkmer-SVM model includes approximately 1 million unique <italic>k</italic>-mers (See <xref ref-type="supplementary-material" rid="supp2">Supplementary file 2C</xref> for full kmer list and weights).</p></sec><sec id="s4-7"><title>Other analysis and data sources</title><p>All genome coordinates from previous mouse genome builds were converted to mm10 using the UCSC liftover tool (<xref ref-type="bibr" rid="bib30">Kuhn et al., 2013</xref>). Binding matrices for SOX2, OCT4, KLF4, ESRRB were as previously reported (<xref ref-type="bibr" rid="bib14">Fiore and Cohen, 2016</xref>). The Bedtools suite (version 2.20) was used for manipulations and analysis of bed files (<xref ref-type="bibr" rid="bib43">Quinlan and Hall, 2010</xref>). Statistical tests were chosen based on expectations of normalcy, with Wilcoxon rank-sum test used for comparisons of BC expression as these distributions were observed to be skewed for some library members, Welch’s t-test used where sample sizes were equal and roughly normal, and Fisher’s 1-sided tests used for testing for enrichment in small sample sizes.</p></sec><sec id="s4-8"><title>Data access</title><p>Raw sequencing data for SYN library and gWT/gMUT library can be found under SRA accession number SRR7515851. Processed sequencing data, specifically demultiplexed barcode counts per replicate, can be found under GEO accession number GSE120240. Additionally, a table of normalized reads per million (RPMs) across replicates for all barcodes are included as <xref ref-type="supplementary-material" rid="supp1">Supplementary file 1D</xref> for the SYN library, <xref ref-type="supplementary-material" rid="supp1">Supplementary file 1I</xref> for the gWT/gMUT library, and <xref ref-type="supplementary-material" rid="supp1">Supplementary file 1M</xref> for the MiniSpacer library.</p></sec></sec></body><back><ack id="ack"><title>Acknowledgements</title><p>We thank members of the Cohen Lab for critical reading and feedback, particularly Michael White, Max Staller and Hemangi Chaudhari for helpful discussion over the course of the project, Jessica Hoisington-Lopez from the DNA Sequencing Innovation Lab for assistance with high-throughput sequencing, and Karl Kumbier for modeling discussions. This work is supported by a grant from the National Institutes of Health, R01 GM092910 to BAC.</p></ack><sec id="s5" sec-type="additional-information"><title>Additional information</title><fn-group content-type="competing-interest"><title>Competing interests</title><fn fn-type="COI-statement" id="conf1"><p>No competing interests declared</p></fn></fn-group><fn-group content-type="author-contribution"><title>Author contributions</title><fn fn-type="con" id="con1"><p>Conceptualization, Data curation, Formal analysis, Investigation, Visualization, Methodology</p></fn><fn fn-type="con" id="con2"><p>Data curation, Formal analysis, Methodology, Writing - review and editing</p></fn><fn fn-type="con" id="con3"><p>Investigation, Writing - review and editing</p></fn><fn fn-type="con" id="con4"><p>Investigation</p></fn><fn fn-type="con" id="con5"><p>Data curation, Formal analysis, Methodology</p></fn><fn fn-type="con" id="con6"><p>Conceptualization, Supervision, Funding acquisition</p></fn></fn-group></sec><sec id="s6" sec-type="supplementary-material"><title>Additional files</title><supplementary-material id="supp1"><label>Supplementary file 1.</label><caption><title>Composition of libraries and expression measurements.</title></caption><media mime-subtype="xlsx" mimetype="application" xlink:href="elife-41279-supp1-v2.xlsx"/></supplementary-material><supplementary-material id="supp2"><label>Supplementary file 2.</label><caption><title>Features, weights and motifs of iRF and gkmSVM models.</title></caption><media mime-subtype="xlsx" mimetype="application" xlink:href="elife-41279-supp2-v2.xlsx"/></supplementary-material><supplementary-material id="supp3"><label>Supplementary file 3.</label><caption><title>Sequences of primers used in this study.</title></caption><media mime-subtype="xlsx" mimetype="application" xlink:href="elife-41279-supp3-v2.xlsx"/></supplementary-material><supplementary-material id="supp4"><label>Supplementary file 4.</label><caption><title>FASTA-format input file of GEN library sequences for gkmSVM.</title></caption><media mime-subtype="octet-stream" mimetype="application" xlink:href="elife-41279-supp4-v2.fasta"/></supplementary-material><supplementary-material id="transrepform"><label>Transparent reporting form</label><media mime-subtype="docx" mimetype="application" xlink:href="elife-41279-transrepform-v2.docx"/></supplementary-material></sec><sec id="s7" sec-type="data-availability"><title>Data availability</title><p>Sequencing data has been deposited in GEO under accession code GSE120240. Any additional data generated during this study are included in the manuscript and supporting files.</p><p>The following dataset was generated:</p><p><element-citation id="dataset1" publication-type="data" specific-use="isSupplementedBy"><person-group person-group-type="author"><name><surname>King</surname><given-names>DM</given-names></name><name><surname>Maricque</surname><given-names>BB</given-names></name><name><surname>Cohen</surname><given-names>BA</given-names></name></person-group><year iso-8601-date="2019">2019</year><data-title>Massively Parallel Reporter Assay for pluripotency factors in mESCs</data-title><source>NCBI Gene Expression Omnibus</source><pub-id assigning-authority="NCBI" pub-id-type="accession" xlink:href="https://www.ncbi.nlm.nih.gov/geo/query/acc.cgi?acc=GSE120240">GSE120240</pub-id></element-citation></p><p>The following previously published datasets were used:</p><p><element-citation id="dataset2" publication-type="data" specific-use="references"><person-group person-group-type="author"><name><surname>Chen</surname><given-names>X</given-names></name><name><surname>Xu</surname><given-names>H</given-names></name><name><surname>Yuan</surname><given-names>P</given-names></name><name><surname>Fang</surname><given-names>F</given-names></name><name><surname>Huss</surname><given-names>M</given-names></name><name><surname>Vega</surname><given-names>VB</given-names></name><name><surname>Wong</surname><given-names>E</given-names></name><name><surname>Orlov</surname><given-names>YL</given-names></name><name><surname>Zhang</surname><given-names>W</given-names></name><name><surname>Jiang</surname><given-names>J</given-names></name><name><surname>Loh</surname><given-names>YH</given-names></name><name><surname>Yeo</surname><given-names>HC</given-names></name><name><surname>Yeo</surname><given-names>ZX</given-names></name><name><surname>Narang</surname><given-names>V</given-names></name><name><surname>Govindarajan</surname><given-names>KR</given-names></name><name><surname>Leong</surname><given-names>B</given-names></name><name><surname>Shahab</surname><given-names>A</given-names></name><name><surname>Ruan</surname><given-names>Y</given-names></name><name><surname>Bourque</surname><given-names>G</given-names></name><name><surname>Sung</surname><given-names>WK</given-names></name><name><surname>Clarke</surname><given-names>ND</given-names></name><name><surname>Wei</surname><given-names>CL</given-names></name><name><surname>Ng</surname><given-names>HH</given-names></name></person-group><year iso-8601-date="2008">2008</year><data-title>Mapping of transcription factor binding sites in mouse embryonic stem cells</data-title><source>NCBI Gene Expression Omnibus</source><pub-id assigning-authority="NCBI" pub-id-type="accession" xlink:href="https://www.ncbi.nlm.nih.gov/geo/query/acc.cgi?acc=GSE11431">GSE11431</pub-id></element-citation></p><p><element-citation id="dataset3" publication-type="data" specific-use="references"><person-group person-group-type="author"><name><surname>Yu</surname><given-names>HB</given-names></name><name><surname>Johnson</surname><given-names>R</given-names></name><name><surname>Kunarso</surname><given-names>G</given-names></name><name><surname>Stanton</surname><given-names>LW</given-names></name></person-group><year iso-8601-date="2011">2011</year><data-title>Genome-wide maps of REST and its cofactors in mouse E14 cells</data-title><source>NCBI Gene Expression Omnibus</source><pub-id assigning-authority="NCBI" pub-id-type="accession" xlink:href="https://www.ncbi.nlm.nih.gov/geo/query/acc.cgi?acc=GSE28233">GSE28233</pub-id></element-citation></p><p><element-citation id="dataset4" publication-type="data" specific-use="references"><person-group person-group-type="author"><name><surname>Perino</surname><given-names>M</given-names></name><name><surname>van</surname><given-names>Mierlo G</given-names></name><name><surname>Karemaker</surname><given-names>ID</given-names></name><name><surname>van</surname><given-names>Genesen S</given-names></name><name><surname>Vermeulen</surname><given-names>M</given-names></name><name><surname>Marks</surname><given-names>H</given-names></name><name><surname>van</surname><given-names>Heeringen SJ</given-names></name><name><surname>Veenstra</surname><given-names>GJC</given-names></name></person-group><year iso-8601-date="2018">2018</year><data-title>MTF2 recruits Polycomb Repressive Complex 2 by helical shape-selective DNA binding</data-title><source>NCBI Gene Expression Omnibus</source><pub-id assigning-authority="NCBI" pub-id-type="accession" xlink:href="https://www.ncbi.nlm.nih.gov/geo/query/acc.cgi?acc=GSE94300">GSE94300</pub-id></element-citation></p><p><element-citation id="dataset5" publication-type="data" specific-use="references"><person-group person-group-type="author"><name><surname>Wu</surname><given-names>J</given-names></name><name><surname>Huang</surname><given-names>B</given-names></name><name><surname>Chen</surname><given-names>H</given-names></name><name><surname>Yin</surname><given-names>Q</given-names></name><name><surname>Liu</surname><given-names>Y</given-names></name><name><surname>Xiang</surname><given-names>Y</given-names></name><name><surname>Zhang</surname><given-names>B</given-names></name><name><surname>Liu</surname><given-names>B</given-names></name><name><surname>Wang</surname><given-names>Q</given-names></name><name><surname>Xia</surname><given-names>W</given-names></name><name><surname>Li</surname><given-names>W</given-names></name><name><surname>Li</surname><given-names>Y</given-names></name><name><surname>Ma</surname><given-names>J</given-names></name><name><surname>Peng</surname><given-names>X</given-names></name><name><surname>Zheng</surname><given-names>H</given-names></name><name><surname>Ming</surname><given-names>J</given-names></name><name><surname>Zhang</surname><given-names>W</given-names></name><name><surname>Zhang</surname><given-names>J</given-names></name><name><surname>Tian</surname><given-names>G</given-names></name><name><surname>Xu</surname><given-names>F</given-names></name><name><surname>Chang</surname><given-names>Z</given-names></name><name><surname>Na</surname><given-names>J</given-names></name><name><surname>Yang</surname><given-names>X</given-names></name><name><surname>Xie</surname><given-names>W</given-names></name></person-group><year iso-8601-date="2016">2016</year><data-title>The landscape of accessible chromatin in mammalian pre-implantation embryos (ATAC-Seq)</data-title><source>NCBI Gene Expression Omnibus</source><pub-id assigning-authority="NCBI" pub-id-type="accession" xlink:href="https://www.ncbi.nlm.nih.gov/geo/query/acc.cgi?acc=GSE66581">GSE66581</pub-id></element-citation></p><p><element-citation id="dataset6" publication-type="data" specific-use="references"><person-group person-group-type="author"><name><surname>Pervouchine</surname><given-names>DD</given-names></name><name><surname>Djebali</surname><given-names>S</given-names></name><name><surname>Breschi</surname><given-names>A</given-names></name><name><surname>Davis</surname><given-names>CA</given-names></name><name><surname>Barja</surname><given-names>PP</given-names></name><name><surname>Dobin</surname><given-names>A</given-names></name><name><surname>Tanzer</surname><given-names>A</given-names></name><name><surname>Lagarde</surname><given-names>J</given-names></name><name><surname>Zaleski</surname><given-names>C</given-names></name><name><surname>See</surname><given-names>L</given-names></name><name><surname>Fastuca</surname><given-names>M</given-names></name><name><surname>Drenkow</surname><given-names>J</given-names></name><name><surname>Wang</surname><given-names>H</given-names></name><name><surname>Bussotti</surname><given-names>G</given-names></name><name><surname>Pei</surname><given-names>B</given-names></name><name><surname>Balasubramanian</surname><given-names>S</given-names></name><name><surname>Monlong</surname><given-names>J</given-names></name><name><surname>Harmanci</surname><given-names>A</given-names></name><name><surname>Gerstein</surname><given-names>M</given-names></name><name><surname>Beer</surname><given-names>MA</given-names></name><name><surname>Notredame</surname><given-names>C</given-names></name><name><surname>Guigo</surname><given-names>R</given-names></name><name><surname>Gingeras</surname><given-names>TR</given-names></name></person-group><year iso-8601-date="2014">2014</year><data-title>The conserved organization of the human and mouse transcriptomes</data-title><source>NCBI Gene Expression Omnibus</source><pub-id assigning-authority="NCBI" pub-id-type="accession" xlink:href="https://www.ncbi.nlm.nih.gov/geo/query/acc.cgi?acc=GSE49417">GSE49417</pub-id></element-citation></p></sec><ref-list><title>References</title><ref id="bib1"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Bailey</surname> <given-names>TL</given-names></name><name><surname>Boden</surname> <given-names>M</given-names></name><name><surname>Buske</surname> <given-names>FA</given-names></name><name><surname>Frith</surname> <given-names>M</given-names></name><name><surname>Grant</surname> <given-names>CE</given-names></name><name><surname>Clementi</surname> <given-names>L</given-names></name><name><surname>Ren</surname> <given-names>J</given-names></name><name><surname>Li</surname> <given-names>WW</given-names></name><name><surname>Noble</surname> <given-names>WS</given-names></name></person-group><year iso-8601-date="2009">2009</year><article-title>MEME SUITE: tools for motif discovery and searching</article-title><source>Nucleic Acids Research</source><volume>37</volume><fpage>W202</fpage><lpage>W208</lpage><pub-id pub-id-type="doi">10.1093/nar/gkp335</pub-id><pub-id pub-id-type="pmid">19458158</pub-id></element-citation></ref><ref id="bib2"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Basu</surname> <given-names>S</given-names></name><name><surname>Kumbier</surname> <given-names>K</given-names></name><name><surname>Brown</surname> <given-names>JB</given-names></name><name><surname>Yu</surname> <given-names>B</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Iterative random forests to discover predictive and stable high-order interactions</article-title><source>PNAS</source><volume>115</volume><fpage>1943</fpage><lpage>1948</lpage><pub-id pub-id-type="doi">10.1073/pnas.1711236115</pub-id><pub-id pub-id-type="pmid">29351989</pub-id></element-citation></ref><ref id="bib3"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Chambers</surname> <given-names>I</given-names></name><name><surname>Tomlinson</surname> <given-names>SR</given-names></name></person-group><year iso-8601-date="2009">2009</year><article-title>The transcriptional foundation of pluripotency</article-title><source>Development</source><volume>136</volume><fpage>2311</fpage><lpage>2322</lpage><pub-id pub-id-type="doi">10.1242/dev.024398</pub-id><pub-id pub-id-type="pmid">19542351</pub-id></element-citation></ref><ref id="bib4"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Chaudhari</surname> <given-names>HG</given-names></name><name><surname>Cohen</surname> <given-names>BA</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Local sequence features that influence AP-1 <italic>cis</italic>-regulatory activity</article-title><source>Genome Research</source><volume>28</volume><fpage>171</fpage><lpage>181</lpage><pub-id pub-id-type="doi">10.1101/gr.226530.117</pub-id><pub-id pub-id-type="pmid">29305491</pub-id></element-citation></ref><ref id="bib5"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>CT</given-names></name><name><surname>Gottlieb</surname> <given-names>DI</given-names></name><name><surname>Cohen</surname> <given-names>BA</given-names></name></person-group><year iso-8601-date="2008">2008a</year><article-title>Ultraconserved elements in the Olig2 promoter</article-title><source>PLOS ONE</source><volume>3</volume><elocation-id>e3946</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pone.0003946</pub-id><pub-id pub-id-type="pmid">19079603</pub-id></element-citation></ref><ref id="bib6"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>X</given-names></name><name><surname>Vega</surname> <given-names>VB</given-names></name><name><surname>Ng</surname> <given-names>H-H</given-names></name></person-group><year iso-8601-date="2008">2008b</year><article-title>Transcriptional regulatory networks in embryonic stem cells</article-title><source>Cold Spring Harbor Symposia on Quantitative Biology</source><volume>73</volume><fpage>203</fpage><lpage>209</lpage><pub-id pub-id-type="doi">10.1101/sqb.2008.73.026</pub-id><pub-id pub-id-type="pmid">19022762</pub-id></element-citation></ref><ref id="bib7"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>X</given-names></name><name><surname>Xu</surname> <given-names>H</given-names></name><name><surname>Yuan</surname> <given-names>P</given-names></name><name><surname>Fang</surname> <given-names>F</given-names></name><name><surname>Huss</surname> <given-names>M</given-names></name><name><surname>Vega</surname> <given-names>VB</given-names></name><name><surname>Wong</surname> <given-names>E</given-names></name><name><surname>Orlov</surname> <given-names>YL</given-names></name><name><surname>Zhang</surname> <given-names>W</given-names></name><name><surname>Jiang</surname> <given-names>J</given-names></name><name><surname>Loh</surname> <given-names>YH</given-names></name><name><surname>Yeo</surname> <given-names>HC</given-names></name><name><surname>Yeo</surname> <given-names>ZX</given-names></name><name><surname>Narang</surname> <given-names>V</given-names></name><name><surname>Govindarajan</surname> <given-names>KR</given-names></name><name><surname>Leong</surname> <given-names>B</given-names></name><name><surname>Shahab</surname> <given-names>A</given-names></name><name><surname>Ruan</surname> <given-names>Y</given-names></name><name><surname>Bourque</surname> <given-names>G</given-names></name><name><surname>Sung</surname> <given-names>WK</given-names></name><name><surname>Clarke</surname> <given-names>ND</given-names></name><name><surname>Wei</surname> <given-names>CL</given-names></name><name><surname>Ng</surname> <given-names>HH</given-names></name></person-group><year iso-8601-date="2008">2008c</year><article-title>Integration of external signaling pathways with the core transcriptional network in embryonic stem cells</article-title><source>Cell</source><volume>133</volume><fpage>1106</fpage><lpage>1117</lpage><pub-id pub-id-type="doi">10.1016/j.cell.2008.04.043</pub-id><pub-id pub-id-type="pmid">18555785</pub-id></element-citation></ref><ref id="bib8"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname> <given-names>CY</given-names></name><name><surname>Morris</surname> <given-names>Q</given-names></name><name><surname>Mitchell</surname> <given-names>JA</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>Enhancer identification in mouse embryonic stem cells using integrative modeling of chromatin and genomic features</article-title><source>BMC Genomics</source><volume>13</volume><elocation-id>152</elocation-id><pub-id pub-id-type="doi">10.1186/1471-2164-13-152</pub-id><pub-id pub-id-type="pmid">22537144</pub-id></element-citation></ref><ref id="bib9"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Cruz-Molina</surname> <given-names>S</given-names></name><name><surname>Respuela</surname> <given-names>P</given-names></name><name><surname>Tebartz</surname> <given-names>C</given-names></name><name><surname>Kolovos</surname> <given-names>P</given-names></name><name><surname>Nikolic</surname> <given-names>M</given-names></name><name><surname>Fueyo</surname> <given-names>R</given-names></name><name><surname>van Ijcken</surname> <given-names>WFJ</given-names></name><name><surname>Grosveld</surname> <given-names>F</given-names></name><name><surname>Frommolt</surname> <given-names>P</given-names></name><name><surname>Bazzi</surname> <given-names>H</given-names></name><name><surname>Rada-Iglesias</surname> <given-names>A</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>PRC2 facilitates the regulatory topology required for poised enhancer function during pluripotent stem cell differentiation</article-title><source>Cell Stem Cell</source><volume>20</volume><fpage>689</fpage><lpage>705</lpage><pub-id pub-id-type="doi">10.1016/j.stem.2017.02.004</pub-id><pub-id pub-id-type="pmid">28285903</pub-id></element-citation></ref><ref id="bib10"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Dunn</surname> <given-names>SJ</given-names></name><name><surname>Martello</surname> <given-names>G</given-names></name><name><surname>Yordanov</surname> <given-names>B</given-names></name><name><surname>Emmott</surname> <given-names>S</given-names></name><name><surname>Smith</surname> <given-names>AG</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>Defining an essential transcription factor program for naïve pluripotency</article-title><source>Science</source><volume>344</volume><fpage>1156</fpage><lpage>1160</lpage><pub-id pub-id-type="doi">10.1126/science.1248882</pub-id><pub-id pub-id-type="pmid">24904165</pub-id></element-citation></ref><ref id="bib11"><element-citation publication-type="journal"><person-group person-group-type="author"><collab>ENCODE Project Consortium</collab></person-group><year iso-8601-date="2012">2012</year><article-title>An integrated encyclopedia of DNA elements in the human genome</article-title><source>Nature</source><volume>489</volume><fpage>57</fpage><lpage>74</lpage><pub-id pub-id-type="doi">10.1038/nature11247</pub-id><pub-id pub-id-type="pmid">22955616</pub-id></element-citation></ref><ref id="bib12"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Evans</surname> <given-names>NC</given-names></name><name><surname>Swanson</surname> <given-names>CI</given-names></name><name><surname>Barolo</surname> <given-names>S</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>Sparkling insights into enhancer structure, function, and evolution</article-title><source>Current Topics in Developmental Biology</source><volume>98</volume><fpage>97</fpage><lpage>120</lpage><pub-id pub-id-type="doi">10.1016/B978-0-12-386499-4.00004-5</pub-id><pub-id pub-id-type="pmid">22305160</pub-id></element-citation></ref><ref id="bib13"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Feng</surname> <given-names>B</given-names></name><name><surname>Jiang</surname> <given-names>J</given-names></name><name><surname>Kraus</surname> <given-names>P</given-names></name><name><surname>Ng</surname> <given-names>JH</given-names></name><name><surname>Heng</surname> <given-names>JC</given-names></name><name><surname>Chan</surname> <given-names>YS</given-names></name><name><surname>Yaw</surname> <given-names>LP</given-names></name><name><surname>Zhang</surname> <given-names>W</given-names></name><name><surname>Loh</surname> <given-names>YH</given-names></name><name><surname>Han</surname> <given-names>J</given-names></name><name><surname>Vega</surname> <given-names>VB</given-names></name><name><surname>Cacheux-Rataboul</surname> <given-names>V</given-names></name><name><surname>Lim</surname> <given-names>B</given-names></name><name><surname>Lufkin</surname> <given-names>T</given-names></name><name><surname>Ng</surname> <given-names>HH</given-names></name></person-group><year iso-8601-date="2009">2009</year><article-title>Reprogramming of fibroblasts into induced pluripotent stem cells with orphan nuclear receptor esrrb</article-title><source>Nature Cell Biology</source><volume>11</volume><fpage>197</fpage><lpage>203</lpage><pub-id pub-id-type="doi">10.1038/ncb1827</pub-id><pub-id pub-id-type="pmid">19136965</pub-id></element-citation></ref><ref id="bib14"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Fiore</surname> <given-names>C</given-names></name><name><surname>Cohen</surname> <given-names>BA</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Interactions between pluripotency factors specify <italic>cis</italic>-regulation in embryonic stem cells</article-title><source>Genome Research</source><volume>26</volume><fpage>778</fpage><lpage>786</lpage><pub-id pub-id-type="doi">10.1101/gr.200733.115</pub-id><pub-id pub-id-type="pmid">27197208</pub-id></element-citation></ref><ref id="bib15"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Fisher</surname> <given-names>WW</given-names></name><name><surname>Li</surname> <given-names>JJ</given-names></name><name><surname>Hammonds</surname> <given-names>AS</given-names></name><name><surname>Brown</surname> <given-names>JB</given-names></name><name><surname>Pfeiffer</surname> <given-names>BD</given-names></name><name><surname>Weiszmann</surname> <given-names>R</given-names></name><name><surname>MacArthur</surname> <given-names>S</given-names></name><name><surname>Thomas</surname> <given-names>S</given-names></name><name><surname>Stamatoyannopoulos</surname> <given-names>JA</given-names></name><name><surname>Eisen</surname> <given-names>MB</given-names></name><name><surname>Bickel</surname> <given-names>PJ</given-names></name><name><surname>Biggin</surname> <given-names>MD</given-names></name><name><surname>Celniker</surname> <given-names>SE</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>DNA regions bound at low occupancy by transcription factors do not drive patterned reporter gene expression in <italic>Drosophila</italic></article-title><source>PNAS</source><volume>109</volume><fpage>21330</fpage><lpage>21335</lpage><pub-id pub-id-type="doi">10.1073/pnas.1209589110</pub-id><pub-id pub-id-type="pmid">23236164</pub-id></element-citation></ref><ref id="bib16"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Fletez-Brant</surname> <given-names>C</given-names></name><name><surname>Lee</surname> <given-names>D</given-names></name><name><surname>McCallion</surname> <given-names>AS</given-names></name><name><surname>Beer</surname> <given-names>MA</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>kmer-SVM: a web server for identifying predictive regulatory sequence features in genomic data sets</article-title><source>Nucleic Acids Research</source><volume>41</volume><fpage>W544</fpage><lpage>W556</lpage><pub-id pub-id-type="doi">10.1093/nar/gkt519</pub-id><pub-id pub-id-type="pmid">23771147</pub-id></element-citation></ref><ref id="bib17"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Gabut</surname> <given-names>M</given-names></name><name><surname>Samavarchi-Tehrani</surname> <given-names>P</given-names></name><name><surname>Wang</surname> <given-names>X</given-names></name><name><surname>Slobodeniuc</surname> <given-names>V</given-names></name><name><surname>O'Hanlon</surname> <given-names>D</given-names></name><name><surname>Sung</surname> <given-names>HK</given-names></name><name><surname>Alvarez</surname> <given-names>M</given-names></name><name><surname>Talukder</surname> <given-names>S</given-names></name><name><surname>Pan</surname> <given-names>Q</given-names></name><name><surname>Mazzoni</surname> <given-names>EO</given-names></name><name><surname>Nedelec</surname> <given-names>S</given-names></name><name><surname>Wichterle</surname> <given-names>H</given-names></name><name><surname>Woltjen</surname> <given-names>K</given-names></name><name><surname>Hughes</surname> <given-names>TR</given-names></name><name><surname>Zandstra</surname> <given-names>PW</given-names></name><name><surname>Nagy</surname> <given-names>A</given-names></name><name><surname>Wrana</surname> <given-names>JL</given-names></name><name><surname>Blencowe</surname> <given-names>BJ</given-names></name></person-group><year iso-8601-date="2011">2011</year><article-title>An alternative splicing switch regulates embryonic stem cell pluripotency and reprogramming</article-title><source>Cell</source><volume>147</volume><fpage>132</fpage><lpage>146</lpage><pub-id pub-id-type="doi">10.1016/j.cell.2011.08.023</pub-id><pub-id pub-id-type="pmid">21924763</pub-id></element-citation></ref><ref id="bib18"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ghandi</surname> <given-names>M</given-names></name><name><surname>Lee</surname> <given-names>D</given-names></name><name><surname>Mohammad-Noori</surname> <given-names>M</given-names></name><name><surname>Beer</surname> <given-names>MA</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>Enhanced regulatory sequence prediction using gapped k-mer features</article-title><source>PLOS Computational Biology</source><volume>10</volume><elocation-id>e1003711</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pcbi.1003711</pub-id><pub-id pub-id-type="pmid">25033408</pub-id></element-citation></ref><ref id="bib19"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ghandi</surname> <given-names>M</given-names></name><name><surname>Mohammad-Noori</surname> <given-names>M</given-names></name><name><surname>Ghareghani</surname> <given-names>N</given-names></name><name><surname>Lee</surname> <given-names>D</given-names></name><name><surname>Garraway</surname> <given-names>L</given-names></name><name><surname>Beer</surname> <given-names>MA</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>gkmSVM: an R package for gapped-kmer SVM</article-title><source>Bioinformatics</source><volume>32</volume><fpage>2205</fpage><lpage>2207</lpage><pub-id pub-id-type="doi">10.1093/bioinformatics/btw203</pub-id><pub-id pub-id-type="pmid">27153639</pub-id></element-citation></ref><ref id="bib20"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Giorgetti</surname> <given-names>L</given-names></name><name><surname>Siggers</surname> <given-names>T</given-names></name><name><surname>Tiana</surname> <given-names>G</given-names></name><name><surname>Caprara</surname> <given-names>G</given-names></name><name><surname>Notarbartolo</surname> <given-names>S</given-names></name><name><surname>Corona</surname> <given-names>T</given-names></name><name><surname>Pasparakis</surname> <given-names>M</given-names></name><name><surname>Milani</surname> <given-names>P</given-names></name><name><surname>Bulyk</surname> <given-names>ML</given-names></name><name><surname>Natoli</surname> <given-names>G</given-names></name></person-group><year iso-8601-date="2010">2010</year><article-title>Noncooperative interactions between transcription factors and clustered DNA binding sites enable graded transcriptional responses to environmental inputs</article-title><source>Molecular Cell</source><volume>37</volume><fpage>418</fpage><lpage>428</lpage><pub-id pub-id-type="doi">10.1016/j.molcel.2010.01.016</pub-id><pub-id pub-id-type="pmid">20159560</pub-id></element-citation></ref><ref id="bib21"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Grant</surname> <given-names>CE</given-names></name><name><surname>Bailey</surname> <given-names>TL</given-names></name><name><surname>Noble</surname> <given-names>WS</given-names></name></person-group><year iso-8601-date="2011">2011</year><article-title>FIMO: scanning for occurrences of a given motif</article-title><source>Bioinformatics</source><volume>27</volume><fpage>1017</fpage><lpage>1018</lpage><pub-id pub-id-type="doi">10.1093/bioinformatics/btr064</pub-id><pub-id pub-id-type="pmid">21330290</pub-id></element-citation></ref><ref id="bib22"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Grossman</surname> <given-names>SR</given-names></name><name><surname>Zhang</surname> <given-names>X</given-names></name><name><surname>Wang</surname> <given-names>L</given-names></name><name><surname>Engreitz</surname> <given-names>J</given-names></name><name><surname>Melnikov</surname> <given-names>A</given-names></name><name><surname>Rogov</surname> <given-names>P</given-names></name><name><surname>Tewhey</surname> <given-names>R</given-names></name><name><surname>Isakova</surname> <given-names>A</given-names></name><name><surname>Deplancke</surname> <given-names>B</given-names></name><name><surname>Bernstein</surname> <given-names>BE</given-names></name><name><surname>Mikkelsen</surname> <given-names>TS</given-names></name><name><surname>Lander</surname> <given-names>ES</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>Systematic dissection of genomic features determining transcription factor binding and enhancer function</article-title><source>PNAS</source><volume>114</volume><fpage>E1291</fpage><lpage>E1300</lpage><pub-id pub-id-type="doi">10.1073/pnas.1621150114</pub-id><pub-id pub-id-type="pmid">28137873</pub-id></element-citation></ref><ref id="bib23"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hare</surname> <given-names>EE</given-names></name><name><surname>Peterson</surname> <given-names>BK</given-names></name><name><surname>Eisen</surname> <given-names>MB</given-names></name></person-group><year iso-8601-date="2008">2008a</year><article-title>A careful look at binding site reorganization in the even-skipped enhancers of <italic>Drosophila</italic> and sepsids</article-title><source>PLOS Genetics</source><volume>4</volume><elocation-id>e1000268</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pgen.1000268</pub-id></element-citation></ref><ref id="bib24"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hare</surname> <given-names>EE</given-names></name><name><surname>Peterson</surname> <given-names>BK</given-names></name><name><surname>Iyer</surname> <given-names>VN</given-names></name><name><surname>Meier</surname> <given-names>R</given-names></name><name><surname>Eisen</surname> <given-names>MB</given-names></name></person-group><year iso-8601-date="2008">2008b</year><article-title>Sepsid even-skipped enhancers are functionally conserved in <italic>Drosophila</italic> despite lack of sequence conservation</article-title><source>PLOS Genetics</source><volume>4</volume><elocation-id>e1000106</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pgen.1000106</pub-id><pub-id pub-id-type="pmid">18584029</pub-id></element-citation></ref><ref id="bib25"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Huang</surname> <given-names>J</given-names></name><name><surname>Chen</surname> <given-names>T</given-names></name><name><surname>Liu</surname> <given-names>X</given-names></name><name><surname>Jiang</surname> <given-names>J</given-names></name><name><surname>Li</surname> <given-names>J</given-names></name><name><surname>Li</surname> <given-names>D</given-names></name><name><surname>Liu</surname> <given-names>XS</given-names></name><name><surname>Li</surname> <given-names>W</given-names></name><name><surname>Kang</surname> <given-names>J</given-names></name><name><surname>Pei</surname> <given-names>G</given-names></name></person-group><year iso-8601-date="2009">2009</year><article-title>More synergetic cooperation of Yamanaka factors in induced pluripotent stem cells than in embryonic stem cells</article-title><source>Cell Research</source><volume>19</volume><fpage>1127</fpage><lpage>1138</lpage><pub-id pub-id-type="doi">10.1038/cr.2009.106</pub-id><pub-id pub-id-type="pmid">19736564</pub-id></element-citation></ref><ref id="bib26"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Inoue</surname> <given-names>F</given-names></name><name><surname>Kircher</surname> <given-names>M</given-names></name><name><surname>Martin</surname> <given-names>B</given-names></name><name><surname>Cooper</surname> <given-names>GM</given-names></name><name><surname>Witten</surname> <given-names>DM</given-names></name><name><surname>McManus</surname> <given-names>MT</given-names></name><name><surname>Ahituv</surname> <given-names>N</given-names></name><name><surname>Shendure</surname> <given-names>J</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>A systematic comparison reveals substantial differences in chromosomal versus episomal encoding of enhancer activity</article-title><source>Genome Research</source><volume>27</volume><fpage>38</fpage><lpage>52</lpage><pub-id pub-id-type="doi">10.1101/gr.212092.116</pub-id><pub-id pub-id-type="pmid">27831498</pub-id></element-citation></ref><ref id="bib27"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Jauch</surname> <given-names>R</given-names></name><name><surname>Ng</surname> <given-names>CK</given-names></name><name><surname>Saikatendu</surname> <given-names>KS</given-names></name><name><surname>Stevens</surname> <given-names>RC</given-names></name><name><surname>Kolatkar</surname> <given-names>PR</given-names></name></person-group><year iso-8601-date="2008">2008</year><article-title>Crystal structure and DNA binding of the homeodomain of the stem cell transcription factor nanog</article-title><source>Journal of Molecular Biology</source><volume>376</volume><fpage>758</fpage><lpage>770</lpage><pub-id pub-id-type="doi">10.1016/j.jmb.2007.11.091</pub-id><pub-id pub-id-type="pmid">18177668</pub-id></element-citation></ref><ref id="bib28"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Junion</surname> <given-names>G</given-names></name><name><surname>Spivakov</surname> <given-names>M</given-names></name><name><surname>Girardot</surname> <given-names>C</given-names></name><name><surname>Braun</surname> <given-names>M</given-names></name><name><surname>Gustafson</surname> <given-names>EH</given-names></name><name><surname>Birney</surname> <given-names>E</given-names></name><name><surname>Furlong</surname> <given-names>EE</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>A transcription factor collective defines cardiac cell fate and reflects lineage history</article-title><source>Cell</source><volume>148</volume><fpage>473</fpage><lpage>486</lpage><pub-id pub-id-type="doi">10.1016/j.cell.2012.01.030</pub-id><pub-id pub-id-type="pmid">22304916</pub-id></element-citation></ref><ref id="bib29"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Khoueiry</surname> <given-names>P</given-names></name><name><surname>Rothbächer</surname> <given-names>U</given-names></name><name><surname>Ohtsuka</surname> <given-names>Y</given-names></name><name><surname>Daian</surname> <given-names>F</given-names></name><name><surname>Frangulian</surname> <given-names>E</given-names></name><name><surname>Roure</surname> <given-names>A</given-names></name><name><surname>Dubchak</surname> <given-names>I</given-names></name><name><surname>Lemaire</surname> <given-names>P</given-names></name></person-group><year iso-8601-date="2010">2010</year><article-title>A cis-regulatory signature in ascidians and flies, independent of transcription factor binding sites</article-title><source>Current Biology</source><volume>20</volume><fpage>792</fpage><lpage>802</lpage><pub-id pub-id-type="doi">10.1016/j.cub.2010.03.063</pub-id><pub-id pub-id-type="pmid">20434338</pub-id></element-citation></ref><ref id="bib30"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kuhn</surname> <given-names>RM</given-names></name><name><surname>Haussler</surname> <given-names>D</given-names></name><name><surname>Kent</surname> <given-names>WJ</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>The UCSC genome browser and associated tools</article-title><source>Briefings in Bioinformatics</source><volume>14</volume><fpage>144</fpage><lpage>161</lpage><pub-id pub-id-type="doi">10.1093/bib/bbs038</pub-id></element-citation></ref><ref id="bib31"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kulkarni</surname> <given-names>MM</given-names></name><name><surname>Arnosti</surname> <given-names>DN</given-names></name></person-group><year iso-8601-date="2003">2003</year><article-title>Information display by transcriptional enhancers</article-title><source>Development</source><volume>130</volume><fpage>6569</fpage><lpage>6575</lpage><pub-id pub-id-type="doi">10.1242/dev.00890</pub-id><pub-id pub-id-type="pmid">14660545</pub-id></element-citation></ref><ref id="bib32"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kwasnieski</surname> <given-names>JC</given-names></name><name><surname>Mogno</surname> <given-names>I</given-names></name><name><surname>Myers</surname> <given-names>CA</given-names></name><name><surname>Corbo</surname> <given-names>JC</given-names></name><name><surname>Cohen</surname> <given-names>BA</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>Complex effects of nucleotide variants in a mammalian cis-regulatory element</article-title><source>PNAS</source><volume>109</volume><fpage>19498</fpage><lpage>19503</lpage><pub-id pub-id-type="doi">10.1073/pnas.1210678109</pub-id></element-citation></ref><ref id="bib33"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname> <given-names>X</given-names></name><name><surname>Huang</surname> <given-names>J</given-names></name><name><surname>Chen</surname> <given-names>T</given-names></name><name><surname>Wang</surname> <given-names>Y</given-names></name><name><surname>Xin</surname> <given-names>S</given-names></name><name><surname>Li</surname> <given-names>J</given-names></name><name><surname>Pei</surname> <given-names>G</given-names></name><name><surname>Kang</surname> <given-names>J</given-names></name></person-group><year iso-8601-date="2008">2008</year><article-title>Yamanaka factors critically regulate the developmental signaling network in mouse embryonic stem cells</article-title><source>Cell Research</source><volume>18</volume><fpage>1177</fpage><lpage>1189</lpage><pub-id pub-id-type="doi">10.1038/cr.2008.309</pub-id><pub-id pub-id-type="pmid">19030024</pub-id></element-citation></ref><ref id="bib34"><element-citation publication-type="confproc"><person-group person-group-type="author"><name><surname>Louppe</surname> <given-names>G</given-names></name><collab>Wehenkel L</collab><collab>Sutera A</collab><collab>Geurts P</collab></person-group><year iso-8601-date="2013">2013</year><article-title>Understanding variable importances in forests of randomized trees</article-title><conf-name>Proceedings of the 26th International Conference on Neural Information Processing Systems</conf-name><fpage>431</fpage><lpage>439</lpage></element-citation></ref><ref id="bib35"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ludwig</surname> <given-names>MZ</given-names></name><name><surname>Bergman</surname> <given-names>C</given-names></name><name><surname>Patel</surname> <given-names>NH</given-names></name><name><surname>Kreitman</surname> <given-names>M</given-names></name></person-group><year iso-8601-date="2000">2000</year><article-title>Evidence for stabilizing selection in a eukaryotic enhancer element</article-title><source>Nature</source><volume>403</volume><fpage>564</fpage><lpage>567</lpage><pub-id pub-id-type="doi">10.1038/35000615</pub-id><pub-id pub-id-type="pmid">10676967</pub-id></element-citation></ref><ref id="bib36"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Maricque</surname> <given-names>BB</given-names></name><name><surname>Chaudhari</surname> <given-names>HG</given-names></name><name><surname>Cohen</surname> <given-names>BA</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>A massively parallel reporter assay dissects the influence of chromatin structure on cis-regulatory activity</article-title><source>Nature Biotechnology</source><volume>37</volume><fpage>90</fpage><lpage>95</lpage><pub-id pub-id-type="doi">10.1038/nbt.4285</pub-id></element-citation></ref><ref id="bib37"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Menze</surname> <given-names>BH</given-names></name><name><surname>Kelm</surname> <given-names>BM</given-names></name><name><surname>Masuch</surname> <given-names>R</given-names></name><name><surname>Himmelreich</surname> <given-names>U</given-names></name><name><surname>Bachert</surname> <given-names>P</given-names></name><name><surname>Petrich</surname> <given-names>W</given-names></name><name><surname>Hamprecht</surname> <given-names>FA</given-names></name></person-group><year iso-8601-date="2009">2009</year><article-title>A comparison of random forest and its gini importance with standard chemometric methods for the feature selection and classification of spectral data</article-title><source>BMC Bioinformatics</source><volume>10</volume><elocation-id>213</elocation-id><pub-id pub-id-type="doi">10.1186/1471-2105-10-213</pub-id><pub-id pub-id-type="pmid">19591666</pub-id></element-citation></ref><ref id="bib38"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Mogno</surname> <given-names>I</given-names></name><name><surname>Kwasnieski</surname> <given-names>JC</given-names></name><name><surname>Cohen</surname> <given-names>BA</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>Massively parallel synthetic promoter assays reveal the in vivo effects of binding site variants</article-title><source>Genome Research</source><volume>23</volume><fpage>1908</fpage><lpage>1915</lpage><pub-id pub-id-type="doi">10.1101/gr.157891.113</pub-id><pub-id pub-id-type="pmid">23921661</pub-id></element-citation></ref><ref id="bib39"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Mulas</surname> <given-names>C</given-names></name><name><surname>Chia</surname> <given-names>G</given-names></name><name><surname>Jones</surname> <given-names>KA</given-names></name><name><surname>Hodgson</surname> <given-names>AC</given-names></name><name><surname>Stirparo</surname> <given-names>GG</given-names></name><name><surname>Nichols</surname> <given-names>J</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Oct4 regulates the embryonic Axis and coordinates exit from pluripotency and germ layer specification in the mouse embryo</article-title><source>Development</source><volume>145</volume><elocation-id>dev159103</elocation-id><pub-id pub-id-type="doi">10.1242/dev.159103</pub-id><pub-id pub-id-type="pmid">29915126</pub-id></element-citation></ref><ref id="bib40"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Niwa</surname> <given-names>H</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>The pluripotency transcription factor network at work in reprogramming</article-title><source>Current Opinion in Genetics &amp; Development</source><volume>28</volume><fpage>25</fpage><lpage>31</lpage><pub-id pub-id-type="doi">10.1016/j.gde.2014.08.004</pub-id><pub-id pub-id-type="pmid">25173150</pub-id></element-citation></ref><ref id="bib41"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Pan</surname> <given-names>G</given-names></name><name><surname>Thomson</surname> <given-names>JA</given-names></name></person-group><year iso-8601-date="2007">2007</year><article-title>Nanog and transcriptional networks in embryonic stem cell pluripotency</article-title><source>Cell Research</source><volume>17</volume><fpage>42</fpage><lpage>49</lpage><pub-id pub-id-type="doi">10.1038/sj.cr.7310125</pub-id><pub-id pub-id-type="pmid">17211451</pub-id></element-citation></ref><ref id="bib42"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Panne</surname> <given-names>D</given-names></name></person-group><year iso-8601-date="2008">2008</year><article-title>The enhanceosome</article-title><source>Current Opinion in Structural Biology</source><volume>18</volume><fpage>236</fpage><lpage>242</lpage><pub-id pub-id-type="doi">10.1016/j.sbi.2007.12.002</pub-id><pub-id pub-id-type="pmid">18206362</pub-id></element-citation></ref><ref id="bib43"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Quinlan</surname> <given-names>AR</given-names></name><name><surname>Hall</surname> <given-names>IM</given-names></name></person-group><year iso-8601-date="2010">2010</year><article-title>BEDTools: a flexible suite of utilities for comparing genomic features</article-title><source>Bioinformatics</source><volume>26</volume><fpage>841</fpage><lpage>842</lpage><pub-id pub-id-type="doi">10.1093/bioinformatics/btq033</pub-id><pub-id pub-id-type="pmid">20110278</pub-id></element-citation></ref><ref id="bib44"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Reményi</surname> <given-names>A</given-names></name><name><surname>Lins</surname> <given-names>K</given-names></name><name><surname>Nissen</surname> <given-names>LJ</given-names></name><name><surname>Reinbold</surname> <given-names>R</given-names></name><name><surname>Schöler</surname> <given-names>HR</given-names></name><name><surname>Wilmanns</surname> <given-names>M</given-names></name></person-group><year iso-8601-date="2003">2003</year><article-title>Crystal structure of a POU/HMG/DNA ternary complex suggests differential assembly of Oct4 and Sox2 on two enhancers</article-title><source>Genes &amp; Development</source><volume>17</volume><fpage>2048</fpage><lpage>2059</lpage><pub-id pub-id-type="doi">10.1101/gad.269303</pub-id><pub-id pub-id-type="pmid">12923055</pub-id></element-citation></ref><ref id="bib45"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Reményi</surname> <given-names>A</given-names></name><name><surname>Schöler</surname> <given-names>HR</given-names></name><name><surname>Wilmanns</surname> <given-names>M</given-names></name></person-group><year iso-8601-date="2004">2004</year><article-title>Combinatorial control of gene expression</article-title><source>Nature Structural &amp; Molecular Biology</source><volume>11</volume><fpage>812</fpage><lpage>815</lpage><pub-id pub-id-type="doi">10.1038/nsmb820</pub-id><pub-id pub-id-type="pmid">15332082</pub-id></element-citation></ref><ref id="bib46"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Sandelin</surname> <given-names>A</given-names></name></person-group><year iso-8601-date="2004">2004</year><article-title>JASPAR: an open-access database for eukaryotic transcription factor binding profiles</article-title><source>Nucleic Acids Research</source><volume>32</volume><fpage>91</fpage><lpage>94</lpage><pub-id pub-id-type="doi">10.1093/nar/gkh012</pub-id></element-citation></ref><ref id="bib47"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Segal</surname> <given-names>E</given-names></name><name><surname>Raveh-Sadka</surname> <given-names>T</given-names></name><name><surname>Schroeder</surname> <given-names>M</given-names></name><name><surname>Unnerstall</surname> <given-names>U</given-names></name><name><surname>Gaul</surname> <given-names>U</given-names></name></person-group><year iso-8601-date="2008">2008</year><article-title>Predicting expression patterns from regulatory sequence in <italic>Drosophila</italic> segmentation</article-title><source>Nature</source><volume>451</volume><fpage>535</fpage><lpage>540</lpage><pub-id pub-id-type="doi">10.1038/nature06496</pub-id><pub-id pub-id-type="pmid">18172436</pub-id></element-citation></ref><ref id="bib48"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Singh</surname> <given-names>U</given-names></name><name><surname>Quintanilla</surname> <given-names>RH</given-names></name><name><surname>Grecian</surname> <given-names>S</given-names></name><name><surname>Gee</surname> <given-names>KR</given-names></name><name><surname>Rao</surname> <given-names>MS</given-names></name><name><surname>Lakshmipathy</surname> <given-names>U</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>Novel live alkaline phosphatase substrate for identification of pluripotent stem cells</article-title><source>Stem Cell Reviews and Reports</source><volume>8</volume><fpage>1021</fpage><lpage>1029</lpage><pub-id pub-id-type="doi">10.1007/s12015-012-9359-6</pub-id></element-citation></ref><ref id="bib49"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Spitz</surname> <given-names>F</given-names></name><name><surname>Furlong</surname> <given-names>EE</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>Transcription factors: from enhancer binding to developmental control</article-title><source>Nature Reviews Genetics</source><volume>13</volume><fpage>613</fpage><lpage>626</lpage><pub-id pub-id-type="doi">10.1038/nrg3207</pub-id><pub-id pub-id-type="pmid">22868264</pub-id></element-citation></ref><ref id="bib50"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Takahashi</surname> <given-names>K</given-names></name><name><surname>Yamanaka</surname> <given-names>S</given-names></name></person-group><year iso-8601-date="2006">2006</year><article-title>Induction of pluripotent stem cells from mouse embryonic and adult fibroblast cultures by defined factors</article-title><source>Cell</source><volume>126</volume><fpage>663</fpage><lpage>676</lpage><pub-id pub-id-type="doi">10.1016/j.cell.2006.07.024</pub-id><pub-id pub-id-type="pmid">16904174</pub-id></element-citation></ref><ref id="bib51"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Uhl</surname> <given-names>JD</given-names></name><name><surname>Zandvakili</surname> <given-names>A</given-names></name><name><surname>Gebelein</surname> <given-names>B</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>A hox transcription factor collective binds a highly conserved Distal-less cis-Regulatory module to generate robust transcriptional outcomes</article-title><source>PLOS Genetics</source><volume>12</volume><elocation-id>e1005981</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pgen.1005981</pub-id><pub-id pub-id-type="pmid">27058369</pub-id></element-citation></ref><ref id="bib52"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ulirsch</surname> <given-names>JC</given-names></name><name><surname>Nandakumar</surname> <given-names>SK</given-names></name><name><surname>Wang</surname> <given-names>L</given-names></name><name><surname>Giani</surname> <given-names>FC</given-names></name><name><surname>Zhang</surname> <given-names>X</given-names></name><name><surname>Rogov</surname> <given-names>P</given-names></name><name><surname>Melnikov</surname> <given-names>A</given-names></name><name><surname>McDonel</surname> <given-names>P</given-names></name><name><surname>Do</surname> <given-names>R</given-names></name><name><surname>Mikkelsen</surname> <given-names>TS</given-names></name><name><surname>Sankaran</surname> <given-names>VG</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Systematic functional dissection of common genetic variation affecting red blood cell traits</article-title><source>Cell</source><volume>165</volume><fpage>1530</fpage><lpage>1545</lpage><pub-id pub-id-type="doi">10.1016/j.cell.2016.04.048</pub-id><pub-id pub-id-type="pmid">27259154</pub-id></element-citation></ref><ref id="bib53"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Visel</surname> <given-names>A</given-names></name><name><surname>Blow</surname> <given-names>MJ</given-names></name><name><surname>Li</surname> <given-names>Z</given-names></name><name><surname>Zhang</surname> <given-names>T</given-names></name><name><surname>Akiyama</surname> <given-names>JA</given-names></name><name><surname>Holt</surname> <given-names>A</given-names></name><name><surname>Plajzer-Frick</surname> <given-names>I</given-names></name><name><surname>Shoukry</surname> <given-names>M</given-names></name><name><surname>Wright</surname> <given-names>C</given-names></name><name><surname>Chen</surname> <given-names>F</given-names></name><name><surname>Afzal</surname> <given-names>V</given-names></name><name><surname>Ren</surname> <given-names>B</given-names></name><name><surname>Rubin</surname> <given-names>EM</given-names></name><name><surname>Pennacchio</surname> <given-names>LA</given-names></name></person-group><year iso-8601-date="2009">2009</year><article-title>ChIP-seq accurately predicts tissue-specific activity of enhancers</article-title><source>Nature</source><volume>457</volume><fpage>854</fpage><lpage>858</lpage><pub-id pub-id-type="doi">10.1038/nature07730</pub-id><pub-id pub-id-type="pmid">19212405</pub-id></element-citation></ref><ref id="bib54"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>J</given-names></name><name><surname>Zhuang</surname> <given-names>J</given-names></name><name><surname>Iyer</surname> <given-names>S</given-names></name><name><surname>Lin</surname> <given-names>X</given-names></name><name><surname>Whitfield</surname> <given-names>TW</given-names></name><name><surname>Greven</surname> <given-names>MC</given-names></name><name><surname>Pierce</surname> <given-names>BG</given-names></name><name><surname>Dong</surname> <given-names>X</given-names></name><name><surname>Kundaje</surname> <given-names>A</given-names></name><name><surname>Cheng</surname> <given-names>Y</given-names></name><name><surname>Rando</surname> <given-names>OJ</given-names></name><name><surname>Birney</surname> <given-names>E</given-names></name><name><surname>Myers</surname> <given-names>RM</given-names></name><name><surname>Noble</surname> <given-names>WS</given-names></name><name><surname>Snyder</surname> <given-names>M</given-names></name><name><surname>Weng</surname> <given-names>Z</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>Sequence features and chromatin structure around the genomic regions bound by 119 human transcription factors</article-title><source>Genome Research</source><volume>22</volume><fpage>1798</fpage><lpage>1812</lpage><pub-id pub-id-type="doi">10.1101/gr.139105.112</pub-id></element-citation></ref><ref id="bib55"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname> <given-names>J</given-names></name><name><surname>Zhuang</surname> <given-names>J</given-names></name><name><surname>Iyer</surname> <given-names>S</given-names></name><name><surname>Lin</surname> <given-names>XY</given-names></name><name><surname>Greven</surname> <given-names>MC</given-names></name><name><surname>Kim</surname> <given-names>BH</given-names></name><name><surname>Moore</surname> <given-names>J</given-names></name><name><surname>Pierce</surname> <given-names>BG</given-names></name><name><surname>Dong</surname> <given-names>X</given-names></name><name><surname>Virgil</surname> <given-names>D</given-names></name><name><surname>Birney</surname> <given-names>E</given-names></name><name><surname>Hung</surname> <given-names>JH</given-names></name><name><surname>Weng</surname> <given-names>Z</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>Factorbook.org: a Wiki-based database for transcription factor-binding data generated by the ENCODE consortium</article-title><source>Nucleic Acids Research</source><volume>41</volume><fpage>D171</fpage><lpage>D176</lpage><pub-id pub-id-type="doi">10.1093/nar/gks1221</pub-id><pub-id pub-id-type="pmid">23203885</pub-id></element-citation></ref><ref id="bib56"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Waterston</surname> <given-names>RH</given-names></name><name><surname>Lindblad-Toh</surname> <given-names>K</given-names></name><name><surname>Birney</surname> <given-names>E</given-names></name><name><surname>Rogers</surname> <given-names>J</given-names></name><name><surname>Abril</surname> <given-names>JF</given-names></name><name><surname>Agarwal</surname> <given-names>P</given-names></name><name><surname>Agarwala</surname> <given-names>R</given-names></name><name><surname>Ainscough</surname> <given-names>R</given-names></name><name><surname>Alexandersson</surname> <given-names>M</given-names></name><name><surname>An</surname> <given-names>P</given-names></name><name><surname>Antonarakis</surname> <given-names>SE</given-names></name><name><surname>Attwood</surname> <given-names>J</given-names></name><name><surname>Baertsch</surname> <given-names>R</given-names></name><name><surname>Bailey</surname> <given-names>J</given-names></name><name><surname>Barlow</surname> <given-names>K</given-names></name><name><surname>Beck</surname> <given-names>S</given-names></name><name><surname>Berry</surname> <given-names>E</given-names></name><name><surname>Birren</surname> <given-names>B</given-names></name><name><surname>Bloom</surname> <given-names>T</given-names></name><name><surname>Bork</surname> <given-names>P</given-names></name><name><surname>Botcherby</surname> <given-names>M</given-names></name><name><surname>Bray</surname> <given-names>N</given-names></name><name><surname>Brent</surname> <given-names>MR</given-names></name><name><surname>Brown</surname> <given-names>DG</given-names></name><name><surname>Brown</surname> <given-names>SD</given-names></name><name><surname>Bult</surname> <given-names>C</given-names></name><name><surname>Burton</surname> <given-names>J</given-names></name><name><surname>Butler</surname> <given-names>J</given-names></name><name><surname>Campbell</surname> <given-names>RD</given-names></name><name><surname>Carninci</surname> <given-names>P</given-names></name><name><surname>Cawley</surname> <given-names>S</given-names></name><name><surname>Chiaromonte</surname> <given-names>F</given-names></name><name><surname>Chinwalla</surname> <given-names>AT</given-names></name><name><surname>Church</surname> <given-names>DM</given-names></name><name><surname>Clamp</surname> <given-names>M</given-names></name><name><surname>Clee</surname> <given-names>C</given-names></name><name><surname>Collins</surname> <given-names>FS</given-names></name><name><surname>Cook</surname> <given-names>LL</given-names></name><name><surname>Copley</surname> <given-names>RR</given-names></name><name><surname>Coulson</surname> <given-names>A</given-names></name><name><surname>Couronne</surname> <given-names>O</given-names></name><name><surname>Cuff</surname> <given-names>J</given-names></name><name><surname>Curwen</surname> <given-names>V</given-names></name><name><surname>Cutts</surname> <given-names>T</given-names></name><name><surname>Daly</surname> <given-names>M</given-names></name><name><surname>David</surname> <given-names>R</given-names></name><name><surname>Davies</surname> <given-names>J</given-names></name><name><surname>Delehaunty</surname> <given-names>KD</given-names></name><name><surname>Deri</surname> <given-names>J</given-names></name><name><surname>Dermitzakis</surname> <given-names>ET</given-names></name><name><surname>Dewey</surname> <given-names>C</given-names></name><name><surname>Dickens</surname> <given-names>NJ</given-names></name><name><surname>Diekhans</surname> <given-names>M</given-names></name><name><surname>Dodge</surname> <given-names>S</given-names></name><name><surname>Dubchak</surname> <given-names>I</given-names></name><name><surname>Dunn</surname> <given-names>DM</given-names></name><name><surname>Eddy</surname> <given-names>SR</given-names></name><name><surname>Elnitski</surname> <given-names>L</given-names></name><name><surname>Emes</surname> <given-names>RD</given-names></name><name><surname>Eswara</surname> <given-names>P</given-names></name><name><surname>Eyras</surname> <given-names>E</given-names></name><name><surname>Felsenfeld</surname> <given-names>A</given-names></name><name><surname>Fewell</surname> <given-names>GA</given-names></name><name><surname>Flicek</surname> <given-names>P</given-names></name><name><surname>Foley</surname> <given-names>K</given-names></name><name><surname>Frankel</surname> <given-names>WN</given-names></name><name><surname>Fulton</surname> <given-names>LA</given-names></name><name><surname>Fulton</surname> <given-names>RS</given-names></name><name><surname>Furey</surname> <given-names>TS</given-names></name><name><surname>Gage</surname> <given-names>D</given-names></name><name><surname>Gibbs</surname> <given-names>RA</given-names></name><name><surname>Glusman</surname> <given-names>G</given-names></name><name><surname>Gnerre</surname> <given-names>S</given-names></name><name><surname>Goldman</surname> <given-names>N</given-names></name><name><surname>Goodstadt</surname> <given-names>L</given-names></name><name><surname>Grafham</surname> <given-names>D</given-names></name><name><surname>Graves</surname> <given-names>TA</given-names></name><name><surname>Green</surname> <given-names>ED</given-names></name><name><surname>Gregory</surname> <given-names>S</given-names></name><name><surname>Guigó</surname> <given-names>R</given-names></name><name><surname>Guyer</surname> <given-names>M</given-names></name><name><surname>Hardison</surname> <given-names>RC</given-names></name><name><surname>Haussler</surname> <given-names>D</given-names></name><name><surname>Hayashizaki</surname> <given-names>Y</given-names></name><name><surname>Hillier</surname> <given-names>LW</given-names></name><name><surname>Hinrichs</surname> <given-names>A</given-names></name><name><surname>Hlavina</surname> <given-names>W</given-names></name><name><surname>Holzer</surname> <given-names>T</given-names></name><name><surname>Hsu</surname> <given-names>F</given-names></name><name><surname>Hua</surname> <given-names>A</given-names></name><name><surname>Hubbard</surname> <given-names>T</given-names></name><name><surname>Hunt</surname> <given-names>A</given-names></name><name><surname>Jackson</surname> <given-names>I</given-names></name><name><surname>Jaffe</surname> <given-names>DB</given-names></name><name><surname>Johnson</surname> <given-names>LS</given-names></name><name><surname>Jones</surname> <given-names>M</given-names></name><name><surname>Jones</surname> <given-names>TA</given-names></name><name><surname>Joy</surname> <given-names>A</given-names></name><name><surname>Kamal</surname> <given-names>M</given-names></name><name><surname>Karlsson</surname> <given-names>EK</given-names></name><name><surname>Karolchik</surname> <given-names>D</given-names></name><name><surname>Kasprzyk</surname> <given-names>A</given-names></name><name><surname>Kawai</surname> <given-names>J</given-names></name><name><surname>Keibler</surname> <given-names>E</given-names></name><name><surname>Kells</surname> <given-names>C</given-names></name><name><surname>Kent</surname> <given-names>WJ</given-names></name><name><surname>Kirby</surname> <given-names>A</given-names></name><name><surname>Kolbe</surname> <given-names>DL</given-names></name><name><surname>Korf</surname> <given-names>I</given-names></name><name><surname>Kucherlapati</surname> <given-names>RS</given-names></name><name><surname>Kulbokas</surname> <given-names>EJ</given-names></name><name><surname>Kulp</surname> <given-names>D</given-names></name><name><surname>Landers</surname> <given-names>T</given-names></name><name><surname>Leger</surname> <given-names>JP</given-names></name><name><surname>Leonard</surname> <given-names>S</given-names></name><name><surname>Letunic</surname> <given-names>I</given-names></name><name><surname>Levine</surname> <given-names>R</given-names></name><name><surname>Li</surname> <given-names>J</given-names></name><name><surname>Li</surname> <given-names>M</given-names></name><name><surname>Lloyd</surname> <given-names>C</given-names></name><name><surname>Lucas</surname> <given-names>S</given-names></name><name><surname>Ma</surname> <given-names>B</given-names></name><name><surname>Maglott</surname> <given-names>DR</given-names></name><name><surname>Mardis</surname> <given-names>ER</given-names></name><name><surname>Matthews</surname> <given-names>L</given-names></name><name><surname>Mauceli</surname> <given-names>E</given-names></name><name><surname>Mayer</surname> <given-names>JH</given-names></name><name><surname>McCarthy</surname> <given-names>M</given-names></name><name><surname>McCombie</surname> <given-names>WR</given-names></name><name><surname>McLaren</surname> <given-names>S</given-names></name><name><surname>McLay</surname> <given-names>K</given-names></name><name><surname>McPherson</surname> <given-names>JD</given-names></name><name><surname>Meldrim</surname> <given-names>J</given-names></name><name><surname>Meredith</surname> <given-names>B</given-names></name><name><surname>Mesirov</surname> <given-names>JP</given-names></name><name><surname>Miller</surname> <given-names>W</given-names></name><name><surname>Miner</surname> <given-names>TL</given-names></name><name><surname>Mongin</surname> <given-names>E</given-names></name><name><surname>Montgomery</surname> <given-names>KT</given-names></name><name><surname>Morgan</surname> <given-names>M</given-names></name><name><surname>Mott</surname> <given-names>R</given-names></name><name><surname>Mullikin</surname> <given-names>JC</given-names></name><name><surname>Muzny</surname> <given-names>DM</given-names></name><name><surname>Nash</surname> <given-names>WE</given-names></name><name><surname>Nelson</surname> <given-names>JO</given-names></name><name><surname>Nhan</surname> <given-names>MN</given-names></name><name><surname>Nicol</surname> <given-names>R</given-names></name><name><surname>Ning</surname> <given-names>Z</given-names></name><name><surname>Nusbaum</surname> <given-names>C</given-names></name><name><surname>O'Connor</surname> <given-names>MJ</given-names></name><name><surname>Okazaki</surname> <given-names>Y</given-names></name><name><surname>Oliver</surname> <given-names>K</given-names></name><name><surname>Overton-Larty</surname> <given-names>E</given-names></name><name><surname>Pachter</surname> <given-names>L</given-names></name><name><surname>Parra</surname> <given-names>G</given-names></name><name><surname>Pepin</surname> <given-names>KH</given-names></name><name><surname>Peterson</surname> <given-names>J</given-names></name><name><surname>Pevzner</surname> <given-names>P</given-names></name><name><surname>Plumb</surname> <given-names>R</given-names></name><name><surname>Pohl</surname> <given-names>CS</given-names></name><name><surname>Poliakov</surname> <given-names>A</given-names></name><name><surname>Ponce</surname> <given-names>TC</given-names></name><name><surname>Ponting</surname> <given-names>CP</given-names></name><name><surname>Potter</surname> <given-names>S</given-names></name><name><surname>Quail</surname> <given-names>M</given-names></name><name><surname>Reymond</surname> <given-names>A</given-names></name><name><surname>Roe</surname> <given-names>BA</given-names></name><name><surname>Roskin</surname> <given-names>KM</given-names></name><name><surname>Rubin</surname> <given-names>EM</given-names></name><name><surname>Rust</surname> <given-names>AG</given-names></name><name><surname>Santos</surname> <given-names>R</given-names></name><name><surname>Sapojnikov</surname> <given-names>V</given-names></name><name><surname>Schultz</surname> <given-names>B</given-names></name><name><surname>Schultz</surname> <given-names>J</given-names></name><name><surname>Schwartz</surname> <given-names>MS</given-names></name><name><surname>Schwartz</surname> <given-names>S</given-names></name><name><surname>Scott</surname> <given-names>C</given-names></name><name><surname>Seaman</surname> <given-names>S</given-names></name><name><surname>Searle</surname> <given-names>S</given-names></name><name><surname>Sharpe</surname> <given-names>T</given-names></name><name><surname>Sheridan</surname> <given-names>A</given-names></name><name><surname>Shownkeen</surname> <given-names>R</given-names></name><name><surname>Sims</surname> <given-names>S</given-names></name><name><surname>Singer</surname> <given-names>JB</given-names></name><name><surname>Slater</surname> <given-names>G</given-names></name><name><surname>Smit</surname> <given-names>A</given-names></name><name><surname>Smith</surname> <given-names>DR</given-names></name><name><surname>Spencer</surname> <given-names>B</given-names></name><name><surname>Stabenau</surname> <given-names>A</given-names></name><name><surname>Stange-Thomann</surname> <given-names>N</given-names></name><name><surname>Sugnet</surname> <given-names>C</given-names></name><name><surname>Suyama</surname> <given-names>M</given-names></name><name><surname>Tesler</surname> <given-names>G</given-names></name><name><surname>Thompson</surname> <given-names>J</given-names></name><name><surname>Torrents</surname> <given-names>D</given-names></name><name><surname>Trevaskis</surname> <given-names>E</given-names></name><name><surname>Tromp</surname> <given-names>J</given-names></name><name><surname>Ucla</surname> <given-names>C</given-names></name><name><surname>Ureta-Vidal</surname> <given-names>A</given-names></name><name><surname>Vinson</surname> <given-names>JP</given-names></name><name><surname>Von Niederhausern</surname> <given-names>AC</given-names></name><name><surname>Wade</surname> <given-names>CM</given-names></name><name><surname>Wall</surname> <given-names>M</given-names></name><name><surname>Weber</surname> <given-names>RJ</given-names></name><name><surname>Weiss</surname> <given-names>RB</given-names></name><name><surname>Wendl</surname> <given-names>MC</given-names></name><name><surname>West</surname> <given-names>AP</given-names></name><name><surname>Wetterstrand</surname> <given-names>K</given-names></name><name><surname>Wheeler</surname> <given-names>R</given-names></name><name><surname>Whelan</surname> <given-names>S</given-names></name><name><surname>Wierzbowski</surname> <given-names>J</given-names></name><name><surname>Willey</surname> <given-names>D</given-names></name><name><surname>Williams</surname> <given-names>S</given-names></name><name><surname>Wilson</surname> <given-names>RK</given-names></name><name><surname>Winter</surname> <given-names>E</given-names></name><name><surname>Worley</surname> <given-names>KC</given-names></name><name><surname>Wyman</surname> <given-names>D</given-names></name><name><surname>Yang</surname> <given-names>S</given-names></name><name><surname>Yang</surname> <given-names>SP</given-names></name><name><surname>Zdobnov</surname> <given-names>EM</given-names></name><name><surname>Zody</surname> <given-names>MC</given-names></name><name><surname>Lander</surname> <given-names>ES</given-names></name><collab>Mouse Genome Sequencing Consortium</collab></person-group><year iso-8601-date="2002">2002</year><article-title>Initial sequencing and comparative analysis of the mouse genome</article-title><source>Nature</source><volume>420</volume><fpage>520</fpage><lpage>562</lpage><pub-id pub-id-type="doi">10.1038/nature01262</pub-id><pub-id pub-id-type="pmid">12466850</pub-id></element-citation></ref><ref id="bib57"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>White</surname> <given-names>MA</given-names></name><name><surname>Myers</surname> <given-names>CA</given-names></name><name><surname>Corbo</surname> <given-names>JC</given-names></name><name><surname>Cohen</surname> <given-names>BA</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>Massively parallel in vivo enhancer assay reveals that highly local features determine the cis-regulatory function of ChIP-seq peaks</article-title><source>PNAS</source><volume>110</volume><fpage>11952</fpage><lpage>11957</lpage><pub-id pub-id-type="doi">10.1073/pnas.1307449110</pub-id><pub-id pub-id-type="pmid">23818646</pub-id></element-citation></ref><ref id="bib58"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>White</surname> <given-names>MA</given-names></name><name><surname>Kwasnieski</surname> <given-names>JC</given-names></name><name><surname>Myers</surname> <given-names>CA</given-names></name><name><surname>Shen</surname> <given-names>SQ</given-names></name><name><surname>Corbo</surname> <given-names>JC</given-names></name><name><surname>Cohen</surname> <given-names>BA</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>A simple grammar defines activating and repressing cis-Regulatory elements in photoreceptors</article-title><source>Cell Reports</source><volume>17</volume><fpage>1247</fpage><lpage>1254</lpage><pub-id pub-id-type="doi">10.1016/j.celrep.2016.09.066</pub-id><pub-id pub-id-type="pmid">27783940</pub-id></element-citation></ref><ref id="bib59"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Williams</surname> <given-names>DC</given-names></name><name><surname>Cai</surname> <given-names>M</given-names></name><name><surname>Clore</surname> <given-names>GM</given-names></name></person-group><year iso-8601-date="2004">2004</year><article-title>Molecular basis for synergistic transcriptional activation by Oct1 and Sox2 revealed from the solution structure of the 42-kDa Oct1.Sox2.Hoxb1-DNA ternary transcription factor complex</article-title><source>Journal of Biological Chemistry</source><volume>279</volume><fpage>1449</fpage><lpage>1457</lpage><pub-id pub-id-type="doi">10.1074/jbc.M309790200</pub-id><pub-id pub-id-type="pmid">14559893</pub-id></element-citation></ref><ref id="bib60"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Xian</surname> <given-names>HQ</given-names></name><name><surname>Werth</surname> <given-names>K</given-names></name><name><surname>Gottlieb</surname> <given-names>DI</given-names></name></person-group><year iso-8601-date="2005">2005</year><article-title>Promoter analysis in ES cell-derived neural cells</article-title><source>Biochemical and Biophysical Research Communications</source><volume>327</volume><fpage>155</fpage><lpage>162</lpage><pub-id pub-id-type="doi">10.1016/j.bbrc.2004.11.149</pub-id><pub-id pub-id-type="pmid">15629444</pub-id></element-citation></ref><ref id="bib61"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Yie</surname> <given-names>J</given-names></name><name><surname>Senger</surname> <given-names>K</given-names></name><name><surname>Thanos</surname> <given-names>D</given-names></name></person-group><year iso-8601-date="1999">1999</year><article-title>Mechanism by which the IFN-beta enhanceosome activates transcription</article-title><source>PNAS</source><volume>96</volume><fpage>13108</fpage><lpage>13113</lpage><pub-id pub-id-type="doi">10.1073/pnas.96.23.13108</pub-id><pub-id pub-id-type="pmid">10557281</pub-id></element-citation></ref><ref id="bib62"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname> <given-names>X</given-names></name><name><surname>Zhang</surname> <given-names>J</given-names></name><name><surname>Wang</surname> <given-names>T</given-names></name><name><surname>Esteban</surname> <given-names>MA</given-names></name><name><surname>Pei</surname> <given-names>D</given-names></name></person-group><year iso-8601-date="2008">2008</year><article-title><italic>Esrrb</italic> activates <italic>Oct4</italic> transcription and sustains self-renewal and pluripotency in embryonic stem cells</article-title><source>Journal of Biological Chemistry</source><volume>283</volume><fpage>35825</fpage><lpage>35833</lpage><pub-id pub-id-type="doi">10.1074/jbc.M803481200</pub-id><pub-id pub-id-type="pmid">18957414</pub-id></element-citation></ref><ref id="bib63"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Zhao</surname> <given-names>Y</given-names></name><name><surname>Granas</surname> <given-names>D</given-names></name><name><surname>Stormo</surname> <given-names>GD</given-names></name></person-group><year iso-8601-date="2009">2009</year><article-title>Inferring binding energies from selected binding sites</article-title><source>PLOS Computational Biology</source><volume>5</volume><elocation-id>e1000590</elocation-id><pub-id pub-id-type="doi">10.1371/journal.pcbi.1000590</pub-id><pub-id pub-id-type="pmid">19997485</pub-id></element-citation></ref></ref-list></back><sub-article article-type="decision-letter" id="sa1"><front-stub><article-id pub-id-type="doi">10.7554/eLife.41279.sa1</article-id><title-group><article-title>Decision letter</article-title></title-group><contrib-group><contrib contrib-type="editor"><name><surname>Wittkopp</surname><given-names>Patricia J</given-names></name><role>Reviewing Editor</role><aff><institution>University of Michigan</institution><country>United States</country></aff></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name><surname>Arnosti</surname><given-names>David N</given-names></name><role>Reviewer</role><aff><institution>Michigan State University</institution><country>United States</country></aff></contrib></contrib-group></front-stub><body><boxed-text><p>In the interests of transparency, eLife publishes the most substantive revision requests and the accompanying author responses.</p></boxed-text><p><bold>Acceptance summary:</bold></p><p>We are excited to have this impressive study of <italic>cis</italic>-regulatory grammar published in <italic>eLife</italic>. Figuring out how <italic>cis</italic>-regulatory sequences determine gene expression has been a long-standing challenge for the field, and this study makes an important contribution by revealing the relative roles of transcription factor binding affinity and surrounding sequence context with a rigorous and deep set of experiments.</p><p><bold>Decision letter after peer review:</bold></p><p>Thank you for submitting your article &quot;Synthetic and genomic regulatory elements reveal aspects of <italic>cis</italic>-regulatory grammar in Mouse Embryonic Stem Cells&quot; for consideration by <italic>eLife</italic>. Your article has been reviewed by two peer reviewers, and the evaluation has been overseen by Patricia Wittkopp as both Reviewing and Senior Editor. The following individual involved in review of your submission has agreed to reveal his identity: David N Arnosti (Reviewer #1).</p><p>The reviewers have discussed the reviews with one another and the Reviewing Editor has drafted this decision to help you prepare a revised submission.</p><p>Summary:</p><p>The conversion of DNA sequence information to transcriptional output relies on the context-specific interactions of transcription factors and cofactors, which influence each other and the transcription process in multiple ways. Thus, it has been difficult to identify general principles allowing predictive models of <italic>cis</italic> regulatory element outputs, even with precise information about TF concentrations and DNA sequences acted upon. In this manuscript, King et al. use plasmid-borne massively parallel reporter assays to identify <italic>cis</italic> regulatory considerations that influence the activities of Oct4, Klf4, <italic>Sox2</italic>, and ESRRB transcription factors in embryonic stem cells, which have been well characterized for gene expression and genome-wide chromatin features. They take a two-pronged approach for this study, testing hundreds of elements in plasmid libraries. They create synthetic elements carrying binding motifs for the four proteins in varying numbers (2-4 sites), and they select genomic sequences that resemble these elements in that they are known to have some level of protein occupancy, and bear motifs of the OSKE factors. Using high-throughput sequencing, the activities of the libraries are assessed in transfected cells, and relative expression compared to certain mutant constructs in which the motifs are removed.</p><p>The presence of a large number of inactive elements is a valuable finding that allows the authors to assess the importance of specific features for activity vs. inactivity.</p><p>The authors determined that the activities of the generally active synthetic elements can be predicted by random forest machine learning models, employing the number of bound elements as well as the binding site arrangements. However, the insights gained from these constructs appear not to contain predictive power on the genomic sequences, where a smaller fraction of candidate elements are active. A gapped kmer approach indicates that differentiating the active from inactive genomic sequences involves the identification of additional binding sites. A more nuanced RF modeling approach involving factor spacing, primary sites, and ChIP signals measured on these elements is able to provide a better level of accuracy than any of these elements alone; interestingly, these are factors that were specifically left out of the synthetic library, where spacing is held constant, and motif quality is not varied. Consequently, when using synthetic elements, an enhanceosome model described their data better, while with genomic elements, the billboard model worked better.</p><p>Overall, this study points to possible avenues for progress, as well as very specific reasons for pessimism. The synthetic elements tested include certain features that may not apply to endogenous elements (placement of an element directly next to a basal promoter, plasmid rather than integrated location), as well as consciously avoid variables that may be key (differences in spacing, affinities). Since we already have much evidence for the roles of spacing and motif affinity, it makes sense that the authors deliberately set up a testing situation which can assess other factors for possible use in wider modeling efforts, namely order of elements on the enhancer, and number of factors present. The answer appears to be that any informative synthetic approaches must incorporate the factors pursued in the analysis of their endogenous data elements. Overall, this study makes an important contribution to identification of pathways that must be pursued to subsequently create deeper understanding of the DNA-to-transcriptional output function.</p><p>Essential revisions:</p><p>1) Differing enthusiasm for the modeling component was reflected in the reviewers' comments. One thought that the emphasis on the models was an over-reach, especially because the data neither supports one or the other model, nor is there enough of it to make a definitive claim. The other was not bothered by the data not fitting neatly into either model nor what was perceived as an oversimplification of the models. However, even this latter reviewer agreed that the authors should more clearly spell out how their tests do or do not sample the many variables. Revision to the modeling section to make these points more clear to readers is needed.</p><p>2) There was also a difference in opinion about how much this work advances the field, with one reviewer pointing out that it doesn't identify new physical principles or factors affecting transcriptional regulation, and the other agreeing but arguing that this work is part of the necessary path that our fields must explore to make real progress on predictive approaches. I agree that the large size of the set of active/inactive endogenous elements characterized is a very important contribution to the field; one that will help us better recognize and understand enhancer sequences. Revision to the text to more explicitly articulate the contribution to the field of this work is needed to address this concern.</p><p>Below are specific comments from reviewers that elaborate on the concerns expressed more generally in the two points above:</p><p>1) Even though I like the logo-like presentation of the preferred order of the TF on the synthetic promoters, to claim that this result fits an enhanceosome model is a stretch at best. An enhanceosome model requires positioning of every TF in a particular conserved structure. Here there is a certain preference for some positioning. To my understanding the authors only used identical/constant non-binding site sequence in all of the synthetic constructs. How do they know that these sequences do not influence the order? Can the authors choose two variants (one strong and one weak), and introduce 5 different flanking sequences and check whether the ratio in expression between both arrangements is conserved?</p><p>2) For the genomic elements the results are not surprising, as we expect to get interference from unknown or cryptic regulatory elements. The analysis they provide in Figure 5 shows a nice correlation between ChIP-seq data and strength of expression supporting the notion that the more elements bind the promoter the higher the expression. From my perspective this result supports neither the billboard nor the enhanceosome model, but rather a more dynamic model where the cumulative occupancy keeps the promoter open for longer supporting a higher expression. The dynamic or &quot;occupancy&quot; model is further also supported by the data in Figure 4, where there is a correlation between higher expression and stronger sites. I would like to see a discussion of a more dynamical model as another option for explaining their data.</p><p>3) The claim that optimal spacing leads to better expression is also not supported but the data. First of all – what is optimal spacing? The author don't say. Is it constant for all the TFs used, or different for each TF pair? Do they have data which supports one type of spacing over another? Finally, how do they know that next-nearest neighbor effects do alter &quot;optimal spacing&quot; of nearest neighbors? To answer this question will require another much larger OL, which is clearly outside of the scope. Nevertheless, I would like the authors to clarify their claim.</p><p>4) The authors implicitly factor out a number of elements in their synthetic library, either to make the task manageable, or because they suspect certain features are more important. These include the design of having up to just one binding motif for each of the factors, the placement of the factors adjacent to the basal promoter, ensuring that certain proteins will have privileged access to the basal machinery, and the decision to not test spacing. The logic of the paper would be easier to follow if the authors would explain why they made these choices e.g. perhaps certain features are already well enough known.</p><p>5) The finding that chromatin accessibility is not at all predictive is quite fascinating – many studies have relied on such data to infer where relevant enhancers are, and in which cell types. The authors should place this finding in context – does it have something to do with their use of plasmid-borne genes, rather than integrated reporters?</p><p>6) The RF modeling of genomic sequences with the most complex set of features (58 in all) sorts enhancers into active and inactive elements (if I understood their approach). Would the predictions be different, more informative, if they were attempting to predict relative activity? IF this is a misunderstanding, it would be helpful to clarify.</p></body></sub-article><sub-article article-type="reply" id="sa2"><front-stub><article-id pub-id-type="doi">10.7554/eLife.41279.sa2</article-id><title-group><article-title>Author response</article-title></title-group></front-stub><body><disp-quote content-type="editor-comment"><p>Essential revisions:</p><p>1) Differing enthusiasm for the modeling component was reflected in the reviewers' comments. One thought that the emphasis on the models was an over-reach, especially because the data neither supports one or the other model, nor is there enough of it to make a definitive claim. The other was not bothered by the data not fitting neatly into either model nor what was perceived as an oversimplification of the models. However, even this latter reviewer agreed that the authors should more clearly spell out how their tests do or do not sample the many variables. Revision to the modeling section to make these points more clear to readers is needed.</p></disp-quote><p>In the revised text we removed statements that our data support either the billboard or enhanceosome models because our results do not neatly confirm the predictions of either model. In the revised Introduction we still discuss these two models in order to summarize current thinking in the field [subsection “Regulatory grammar”, first paragraph]. Discussing these models also helps set the stage for the problem our study addresses, which is the extent to which binding sites function independently or through interactions with each other.</p><p>We also clarified which variables are and are not analyzed by the models. We provide justifications for the choices we made in the synthetic library in the section [subsection “Rationale and description of enhancer libraries”, second paragraph.</p><disp-quote content-type="editor-comment"><p>2) There was also a difference in opinion about how much this work advances the field, with one reviewer pointing out that it doesn't identify new physical principles or factors affecting transcriptional regulation, and the other agreeing but arguing that this work is part of the necessary path that our fields must explore to make real progress on predictive approaches. I agree that the large size of the set of active/inactive endogenous elements characterized is a very important contribution to the field; one that will help us better recognize and understand enhancer sequences. Revision to the text to more explicitly articulate the contribution to the field of this work is needed to address this concern.</p></disp-quote><p>In the revised text we attempted to more clearly articulate the contribution our study makes towards a better understanding of gene regulation. We attempted to make three points.</p><p>1) While it is known that enhancers are sequences that contain collections of transcription factor binding, we cannot distinguish true enhancers from spurious conglomerations of binding sites. The most dramatic manifestation of this problem is our presentation of a large number of inactive sequences that have the same sequence features as the active sequences. Distinguishing the properties of these two groups is a major challenge for the field.</p><p>2) While we know that TFs sometimes act independently and other times engage in cooperative interactions, we cannot reliably predict the effects of sequence perturbations to specific binding sites. Our study is an attempt to organize the qualitative principles we know about into a predictive quantitative framework. We were not necessarily looking for new principles of gene regulation. Our rationale is that the principles we already know about will be predictive once they are properly organized into a quantitative framework. Our work is a step towards that framework.</p><p>3) Our study shows that in many cases gene expression is predictable from the sequence of regulatory elements. However, it also shows the large extent to which the principles of gene regulation are context dependent. Our study quantifies the extent to which the context in which binding sites reside influence their activities and formalizes the challenge this will entail.</p><disp-quote content-type="editor-comment"><p>Below are specific comments from reviewers that elaborate on the concerns expressed more generally in the two points above:</p><p>1) Even though I like the logo-like presentation of the preferred order of the TF on the synthetic promoters, to claim that this result fits an enhanceosome model is a stretch at best. An enhanceosome model requires positioning of every TF in a particular conserved structure. Here there is a certain preference for some positioning.</p></disp-quote><p>In the revised text we removed all statements that our data support either the enhanceosome or billboard model since it is true that our data do not clearly rule out either model. The discussion of these models in the Introduction [subsection “Regulatory Grammar”, first paragraph] is meant only as way to bring out current thinking in the field. In the text we have made it more clear that the logos show certain preferences for certain arrangements, but that the results do not rule out any specific model.</p><disp-quote content-type="editor-comment"><p>To my understanding the authors only used identical/constant non-binding site sequence in all of the synthetic constructs. How do they know that these sequences do not influence the order? Can the authors choose two variants (one strong and one weak), and introduce 5 different flanking sequences and check whether the ratio in expression between both arrangements is conserved?</p></disp-quote><p>To address this point we constructed a small library based on six 4-mer synthetic elements in which we systematically tested four new spacer sequences [subsection “Modeling supports a role for TFBS positions in setting expression level for synthetic elements but not for genomic sequences”, last two paragraphs]. The spacer sequences had small effects on expression, with differences ranging from 0.3-25%. However, these small changes were enough to change the rank order of activities of the sequences with each spacer. Our interpretation of this result [Discussion, last paragraph] is that the spacer sequences have small effects on the independent contribution of each transcription factor binding site, which accounts for the overall small effect of spacer sequences, but that different spacer sequences may influence the interactions that occur between binding sites or introduce new interactions with factors that might occupy the spacer sequences themselves. This interpretation is consistent with our observations that the expression of the genomic elements do not correlate well with their corresponding synthetic elements, and supports the hypothesis that sequences other than the binding sites contribute to gene expression.</p><disp-quote content-type="editor-comment"><p>2) For the genomic elements the results are not surprising, as we expect to get interference from unknown or cryptic regulatory elements. The analysis they provide in Figure 5 shows a nice correlation between ChIP-seq data and strength of expression supporting the notion that the more elements bind the promoter the higher the expression. From my perspective this result supports neither the billboard nor the enhanceosome model, but rather a more dynamic model where the cumulative occupancy keeps the promoter open for longer supporting a higher expression. The dynamic or &quot;occupancy&quot; model is further also supported by the data in Figure 4, where there is a correlation between higher expression and stronger sites. I would like to see a discussion of a more dynamical model as another option for explaining their data.</p></disp-quote><p>We have revised the text to increase discussion of the occupancy model [Abstract; subsection “Regulatory Grammar”, last paragraph; Discussion, first two paragraphs]. We have also revised the text throughout the manuscript to focus on the distinction between independence and interaction, rather than on the difference between billboard and enhanceosome.</p><disp-quote content-type="editor-comment"><p>3) The claim that optimal spacing leads to better expression is also not supported but the data. First of all – what is optimal spacing? The author don't say. Is it constant for all the TFs used, or different for each TF pair? Do they have data which supports one type of spacing over another? Finally, how do they know that next-nearest neighbor effects do alter &quot;optimal spacing&quot; of nearest neighbors? To answer this question will require another much larger OL, which is clearly outside of the scope. Nevertheless, I would like the authors to clarify their claim.</p></disp-quote><p>We have removed the claim about optimal spacing. Results demonstrating slight preferences for certain spacings between specific sites in genomic sequences are now presented [subsection “Site affinity contributes to the activity of genomic sequences”, last paragraph and Figure 4—figure supplement 2] and those preferences are incorporated into our final Random Forest model [subsection “Contributions from sites for other transcription factors”, last paragraph].</p><disp-quote content-type="editor-comment"><p>4) The authors implicitly factor out a number of elements in their synthetic library, either to make the task manageable, or because they suspect certain features are more important. These include the design of having up to just one binding motif for each of the factors, the placement of the factors adjacent to the basal promoter, ensuring that certain proteins will have privileged access to the basal machinery, and the decision to not test spacing. The logic of the paper would be easier to follow if the authors would explain why they made these choices e.g. perhaps certain features are already well enough known.</p></disp-quote><p>The text now contains a more detailed discussion of how we chose which parameters to test in the synthetic library [subsection “Rationale and description of enhancer libraries”, second paragraph and subsection “MPRA of reporter gene libraries”].</p><disp-quote content-type="editor-comment"><p>5) The finding that chromatin accessibility is not at all predictive is quite fascinating – many studies have relied on such data to infer where relevant enhancers are, and in which cell types. The authors should place this finding in context – does it have something to do with their use of plasmid-borne genes, rather than integrated reporters?</p></disp-quote><p>The text now discusses several possible explanations for why TF occupancy data, but not DNA accessibility data correlates with activity in our assays [Discussion, second paragraph].</p><disp-quote content-type="editor-comment"><p>6) The RF modeling of genomic sequences with the most complex set of features (58 in all) sorts enhancers into active and inactive elements (if I understood their approach). Would the predictions be different, more informative, if they were attempting to predict relative activity? IF this is a misunderstanding, it would be helpful to clarify.</p></disp-quote><p>This is a subtle point which we now attempt to clarify in the text [subsection “Modeling supports a role for TFBS positions in setting expression level for synthetic elements but not for genomic sequences”, fourth paragraph]. Because 2/3 of the genomic sequences are inactive, the largest signal in the genomic sequences comes from the difference between active and inactive sequences. There is very little power to detect the differences in relative activity among the 1/3 of active genomic sequences. For this reason, when analyzing genomic sequences, we restricted ourselves to models that attempt to distinguish between active and inactive sequences.</p></body></sub-article></article>