<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Archiving and Interchange DTD with MathML3 v1.3 20210610//EN"  "JATS-archivearticle1-3-mathml3.dtd"><article xmlns:ali="http://www.niso.org/schemas/ali/1.0/" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article" dtd-version="1.3"><front><journal-meta><journal-id journal-id-type="nlm-ta">elife</journal-id><journal-id journal-id-type="publisher-id">eLife</journal-id><journal-title-group><journal-title>eLife</journal-title></journal-title-group><issn publication-format="electronic" pub-type="epub">2050-084X</issn><publisher><publisher-name>eLife Sciences Publications, Ltd</publisher-name></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">92991</article-id><article-id pub-id-type="doi">10.7554/eLife.92991</article-id><article-id pub-id-type="doi" specific-use="version">10.7554/eLife.92991.4</article-id><article-version article-version-type="publication-state">version of record</article-version><article-categories><subj-group subj-group-type="display-channel"><subject>Research Article</subject></subj-group><subj-group subj-group-type="heading"><subject>Computational and Systems Biology</subject></subj-group></article-categories><title-group><article-title>Predicting the effect of CRISPR-Cas9-based epigenome editing</article-title></title-group><contrib-group><contrib contrib-type="author" equal-contrib="yes"><name><surname>Batra</surname><given-names>Sanjit Singh</given-names></name><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="fn" rid="equal-contrib1">†</xref><xref ref-type="fn" rid="con1"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author" equal-contrib="yes"><name><surname>Cabrera</surname><given-names>Alan</given-names></name><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="fn" rid="equal-contrib1">†</xref><xref ref-type="fn" rid="con2"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author" equal-contrib="yes"><name><surname>Spence</surname><given-names>Jeffrey P</given-names></name><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0002-3199-1447</contrib-id><xref ref-type="aff" rid="aff3">3</xref><xref ref-type="fn" rid="equal-contrib1">†</xref><xref ref-type="fn" rid="con3"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author"><name><surname>Goell</surname><given-names>Jacob</given-names></name><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="fn" rid="con4"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author"><name><surname>Anand</surname><given-names>Selvalakshmi S</given-names></name><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0009-0000-4436-2618</contrib-id><xref ref-type="aff" rid="aff4">4</xref><xref ref-type="fn" rid="con5"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author" corresp="yes"><name><surname>Hilton</surname><given-names>Isaac B</given-names></name><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0002-3064-8532</contrib-id><email>isaac.hilton@rice.edu</email><xref ref-type="aff" rid="aff2">2</xref><xref ref-type="aff" rid="aff4">4</xref><xref ref-type="other" rid="fund2"/><xref ref-type="fn" rid="con6"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author" corresp="yes"><name><surname>Song</surname><given-names>Yun S</given-names></name><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0002-0734-9868</contrib-id><email>yss@berkeley.edu</email><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="aff" rid="aff5">5</xref><xref ref-type="fn" rid="con7"/><xref ref-type="fn" rid="conf1"/></contrib><aff id="aff1"><label>1</label><institution-wrap><institution-id institution-id-type="ror">https://ror.org/01an7q238</institution-id><institution>Computer Science Division, University of California</institution></institution-wrap><addr-line><named-content content-type="city">Berkeley</named-content></addr-line><country>United States</country></aff><aff id="aff2"><label>2</label><institution-wrap><institution-id institution-id-type="ror">https://ror.org/008zs3103</institution-id><institution>Department of Bioengineering, Rice University</institution></institution-wrap><addr-line><named-content content-type="city">Houston</named-content></addr-line><country>United States</country></aff><aff id="aff3"><label>3</label><institution-wrap><institution-id institution-id-type="ror">https://ror.org/00f54p054</institution-id><institution>Department of Genetics, Stanford University</institution></institution-wrap><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff><aff id="aff4"><label>4</label><institution-wrap><institution-id institution-id-type="ror">https://ror.org/008zs3103</institution-id><institution>Systems, Synthetic, and Physical Biology Graduate Program, Rice University</institution></institution-wrap><addr-line><named-content content-type="city">Houston</named-content></addr-line><country>United States</country></aff><aff id="aff5"><label>5</label><institution-wrap><institution-id institution-id-type="ror">https://ror.org/01an7q238</institution-id><institution>Department of Statistics, University of California, Berkeley</institution></institution-wrap><addr-line><named-content content-type="city">Berkeley</named-content></addr-line><country>United States</country></aff></contrib-group><contrib-group content-type="section"><contrib contrib-type="editor"><name><surname>Khalil</surname><given-names>Ahmad S</given-names></name><role>Reviewing Editor</role><aff><institution-wrap><institution-id institution-id-type="ror">https://ror.org/05qwgg493</institution-id><institution>Boston University</institution></institution-wrap><country>United States</country></aff></contrib><contrib contrib-type="senior_editor"><name><surname>Dalal</surname><given-names>Yamini</given-names></name><role>Senior Editor</role><aff><institution-wrap><institution-id institution-id-type="ror">https://ror.org/040gcmg81</institution-id><institution>National Cancer Institute</institution></institution-wrap><country>United States</country></aff></contrib></contrib-group><author-notes><fn fn-type="con" id="equal-contrib1"><label>†</label><p>These authors contributed equally to this work</p></fn></author-notes><pub-date publication-format="electronic" date-type="publication"><day>12</day><month>01</month><year>2026</year></pub-date><volume>12</volume><elocation-id>RP92991</elocation-id><history><date date-type="sent-for-review" iso-8601-date="2023-10-03"><day>03</day><month>10</month><year>2023</year></date></history><pub-history><event><event-desc>This manuscript was published as a preprint.</event-desc><date date-type="preprint" iso-8601-date="2023-10-03"><day>03</day><month>10</month><year>2023</year></date><self-uri content-type="preprint" xlink:href="https://doi.org/10.1101/2023.10.03.560674"/></event><event><event-desc>This manuscript was published as a reviewed preprint.</event-desc><date date-type="reviewed-preprint" iso-8601-date="2023-12-20"><day>20</day><month>12</month><year>2023</year></date><self-uri content-type="reviewed-preprint" xlink:href="https://doi.org/10.7554/eLife.92991.1"/></event><event><event-desc>The reviewed preprint was revised.</event-desc><date date-type="reviewed-preprint" iso-8601-date="2024-12-19"><day>19</day><month>12</month><year>2024</year></date><self-uri content-type="reviewed-preprint" xlink:href="https://doi.org/10.7554/eLife.92991.2"/></event><event><event-desc>The reviewed preprint was revised.</event-desc><date date-type="reviewed-preprint" iso-8601-date="2025-05-27"><day>27</day><month>05</month><year>2025</year></date><self-uri content-type="reviewed-preprint" xlink:href="https://doi.org/10.7554/eLife.92991.3"/></event></pub-history><permissions><copyright-statement>© 2023, Batra, Cabrera, Spence et al</copyright-statement><copyright-year>2023</copyright-year><copyright-holder>Batra, Cabrera, Spence et al</copyright-holder><ali:free_to_read/><license xlink:href="http://creativecommons.org/licenses/by/4.0/"><ali:license_ref>http://creativecommons.org/licenses/by/4.0/</ali:license_ref><license-p>This article is distributed under the terms of the <ext-link ext-link-type="uri" xlink:href="http://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution License</ext-link>, which permits unrestricted use and redistribution provided that the original author and source are credited.</license-p></license></permissions><self-uri content-type="pdf" xlink:href="elife-92991-v1.pdf"/><self-uri content-type="figures-pdf" xlink:href="elife-92991-figures-v1.pdf"/><abstract><p>Epigenetic regulation orchestrates mammalian transcription, but functional links between them remain elusive. To tackle this problem, we use epigenomic and transcriptomic data from 13 ENCODE cell types to train machine learning models to predict gene expression from histone post-translational modifications (PTMs), achieving transcriptome-wide correlations of ∼0.70−0.79 for most cell types. Our models recapitulate known associations between histone PTMs and expression patterns, including predicting that acetylation of histone subunit H3 lysine residue 27 (H3K27ac) near the transcription start site (TSS) significantly increases expression levels. To validate this prediction experimentally and investigate how natural vs. engineered deposition of H3K27ac might differentially affect expression, we apply the synthetic dCas9-p300 histone acetyltransferase system to 8 genes in the HEK293T cell line and to 5 genes in the K562 cell line. Further, to facilitate model building, we perform MNase-seq to map genome-wide nucleosome occupancy levels in HEK293T. We observe that our models perform well in accurately ranking relative fold-changes among genes in response to the dCas9-p300 system; however, their ability to rank fold-changes within individual genes is noticeably diminished compared to predicting expression across cell types from their native epigenetic signatures. Our findings highlight the need for more comprehensive genome-scale epigenome editing datasets, better understanding of the actual modifications made by epigenome editing tools, and improved causal models that transfer better from endogenous cellular measurements to perturbation experiments. Together, these improvements would facilitate the ability to understand and predictably control the dynamic human epigenome with consequences for human health.</p></abstract><kwd-group kwd-group-type="author-keywords"><kwd>machine learning</kwd><kwd>post-translational modifications</kwd><kwd>gene expression</kwd><kwd>nucleosome occupancy</kwd><kwd>perturbation</kwd></kwd-group><kwd-group kwd-group-type="research-organism"><title>Research organism</title><kwd>Human</kwd></kwd-group><funding-group><award-group id="fund1"><funding-source><institution-wrap><institution-id institution-id-type="ror">https://ror.org/04q48ey07</institution-id><institution>National Institute of General Medical Sciences</institution></institution-wrap></funding-source><award-id>R35-GM134922</award-id><principal-award-recipient><name><surname>Song</surname><given-names>Yun S</given-names></name></principal-award-recipient></award-group><award-group id="fund2"><funding-source><institution-wrap><institution-id institution-id-type="ror">https://ror.org/04q48ey07</institution-id><institution>National Institute of General Medical Sciences</institution></institution-wrap></funding-source><award-id>R35-GM143532</award-id><principal-award-recipient><name><surname>Hilton</surname><given-names>Isaac B</given-names></name></principal-award-recipient></award-group><funding-statement>The funders had no role in study design, data collection and interpretation, or the decision to submit the work for publication.</funding-statement></funding-group><custom-meta-group><custom-meta specific-use="meta-only"><meta-name>Author impact statement</meta-name><meta-value>Machine learning models reveal that histone marks are predictive of gene expression across human cell types and highlight important nuances between natural control and the effects of CRISPR-Cas9-based epigenome editing.</meta-value></custom-meta><custom-meta specific-use="meta-only"><meta-name>publishing-route</meta-name><meta-value>prc</meta-value></custom-meta></custom-meta-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>All cells within a multicellular organism have the same genetic sequence up to a minuscule number of somatic mutations. Yet, many cell types exist with diverse morphological and functional traits. Epigenetics is an important regulator and driver of this diversity by allowing differences in cellular state and gene expression despite having the same genotype (<xref ref-type="bibr" rid="bib67">Taherian Fard and Ragan, 2019</xref>). Indeed, cells traversing the trajectory from pluripotency through terminal differentiation have essentially the same genotype.</p><p>Epigenetic modifications such as post-translational modifications (PTMs) to histone proteins are involved in many vital regulatory processes influencing genomic accessibility, nuclear compartmentalization, and transcription factor binding and recognition (<xref ref-type="bibr" rid="bib50">Reik et al., 2001</xref>; <xref ref-type="bibr" rid="bib34">Kouzarides, 2007</xref>; <xref ref-type="bibr" rid="bib17">Gibney and Nolan, 2010</xref>; <xref ref-type="bibr" rid="bib33">Klemm et al., 2019</xref>; <xref ref-type="bibr" rid="bib21">Hafner and Boettiger, 2023</xref>; <xref ref-type="bibr" rid="bib76">Zhang and Reinberg, 2001</xref>). The Histone Code Hypothesis suggests that combinations of different histone PTMs specify distinct chromatin states, thereby regulating gene expression (<xref ref-type="bibr" rid="bib64">Strahl and Allis, 2000</xref>; <xref ref-type="bibr" rid="bib27">Jenuwein and Allis, 2001</xref>).</p><p>The field of epigenome editing has produced new tools for understanding the outcomes of epigenetic perturbations that promise to be useful for therapeutics by enabling fine-tuned control of gene expression (<xref ref-type="bibr" rid="bib42">Matharu and Ahituv, 2020</xref>; <xref ref-type="bibr" rid="bib68">Thakore et al., 2016</xref>; <xref ref-type="bibr" rid="bib19">Goell and Hilton, 2021</xref>; <xref ref-type="bibr" rid="bib65">Stricker et al., 2017</xref>). Currently, small molecule drugs are used to potently interfere with epigenetic regulation of gene expression. For example, Vorinostat inhibits histone deacetylases, thereby impacting the epigenetic landscape (<xref ref-type="bibr" rid="bib12">Estey, 2013</xref>; <xref ref-type="bibr" rid="bib75">Yoon and Eom, 2016</xref>). However, small molecules globally disrupt the epigenome and transcriptome and therefore are not suitable for targeting individual dysregulated genes nor clarifying epigenetic regulatory mechanisms (<xref ref-type="bibr" rid="bib66">Swaminathan et al., 2007</xref>). Meanwhile, numerous tools have been designed to harness catalytically dead Cas9 (dCas9) to target epigenetic modifiers to DNA sequences encoded in guide RNAs (gRNAs) (<xref ref-type="bibr" rid="bib28">Jinek et al., 2012</xref>; <xref ref-type="bibr" rid="bib41">Mali et al., 2013</xref>; <xref ref-type="bibr" rid="bib23">Hilton et al., 2015</xref>; <xref ref-type="bibr" rid="bib62">Stepper et al., 2017</xref>; <xref ref-type="bibr" rid="bib36">Kwon et al., 2017</xref>; <xref ref-type="bibr" rid="bib37">Li et al., 2021</xref>). CRISPR-Cas9-based epigenome editing strategies facilitate unprecedented, precise control of the epigenome and gene activation, providing a path to epigenetic-based therapeutics (<xref ref-type="bibr" rid="bib6">Cheng et al., 2019</xref>).</p><p>A major challenge for epigenome editing is designing gRNAs that can achieve a desired level of transcriptional or epigenetic modulation. Finding effective gRNAs currently typically requires expensive and low-throughput experimental strategies (<xref ref-type="bibr" rid="bib45">Mohr et al., 2016</xref>; <xref ref-type="bibr" rid="bib38">Liu et al., 2020</xref>; <xref ref-type="bibr" rid="bib39">Mahata et al., 2023</xref>). An alternative approach would be to computationally model how epigenome editing impacts histone PTMs as well as how perturbing these PTMs would consequently impact gene expression.</p><p>To understand how histone PTMs relate to gene expression, large epigenetic and transcriptomic datasets are required. Advancements in high-throughput sequencing have allowed quantification of gene expression and profiling of histone PTMs. Large consortia have performed an extensive number of assays across a wide variety of cell types (<xref ref-type="bibr" rid="bib69">The ENCODE Project Consortium, 2012</xref>; <xref ref-type="bibr" rid="bib35">Kundaje et al., 2015</xref>; <xref ref-type="bibr" rid="bib2">Barrett et al., 2012</xref>).</p><p>These include measurements of histone PTMs, transcription factor binding, gene expression, and chromatin accessibility. These data have enhanced our understanding of how histone PTMs and other chromatin dynamics impact transcriptional regulation (<xref ref-type="bibr" rid="bib30">Keung et al., 2015</xref>; <xref ref-type="bibr" rid="bib49">Rao et al., 2014</xref>; <xref ref-type="bibr" rid="bib24">Holoch and Moazed, 2015</xref>).</p><p>Studying the function of these histone PTMs, however, has been largely limited to statistical associations with gene expression, which may not capture causal relationships (<xref ref-type="bibr" rid="bib29">Karlić et al., 2010</xref>; <xref ref-type="bibr" rid="bib63">Stillman, 2018</xref>; <xref ref-type="bibr" rid="bib60">Singh et al., 2016</xref>). For example, deep learning has been successful in predicting gene expression from epigenetic modifications, such as transcription factor binding (<xref ref-type="bibr" rid="bib52">Schmidt et al., 2017</xref>), chromatin accessibility (<xref ref-type="bibr" rid="bib53">Schmidt et al., 2020</xref>), histone PTMs (<xref ref-type="bibr" rid="bib60">Singh et al., 2016</xref>; <xref ref-type="bibr" rid="bib58">Sekhon et al., 2018</xref>; <xref ref-type="bibr" rid="bib15">Frasca et al., 2022</xref>; <xref ref-type="bibr" rid="bib61">Singh et al., 2017</xref>; <xref ref-type="bibr" rid="bib22">Hamdy et al., 2022</xref>; <xref ref-type="bibr" rid="bib5">Chen et al., 2022</xref>), and DNA methylation (<xref ref-type="bibr" rid="bib79">Zhong et al., 2019</xref>). However, these studies predict gene expression as binary levels instead of a continuous quantity. Finally, as statistical associations can be driven by non-causal mechanisms, it is unclear whether such computational models learn mechanistic, causal relationships between various epigenetic modifications and gene expression. Beyond modeling the relationship between histone PTMs and gene expression, to fully describe how a particular gRNA would affect gene expression, a model of how epigenome editing affects histone PTMs is also required. To our knowledge, there currently are no computational models that can accurately model, in silico, the impact of epigenome editing on histone PTMs.</p><p>Motivated by these observations, we explored models for how epigenome editing impacts histone PTMs as well as how histone PTMs impact gene expression. We used data available through ENCODE (<xref ref-type="bibr" rid="bib55">Schreiber et al., 2020a</xref>; <xref ref-type="bibr" rid="bib69">The ENCODE Project Consortium, 2012</xref>) to train a model of how histone PTMs impact gene expression. Our model is highly predictive of endogenous expression and learns an understanding of chromatin biology which is consistent with known patterns of various histone PTMs (<xref ref-type="bibr" rid="bib31">Kimura, 2013</xref>). To test this model in the context of epigenome editing, we generated perturbation data using the dCas9-p300 histone acetyltransferase system (<xref ref-type="bibr" rid="bib23">Hilton et al., 2015</xref>). The dCas9-p300 system is thought to act primarily through local acetylation of histone lysine residues, particularly histone subunit H3 lysine residue 27 (H3K27ac). Therefore, we modeled the impact of dCas9-p300 on the epigenome as a local increase in the H3K27ac profile near the target site; since the precise effect of these perturbations is unknown, we tried a variety of potential modification patterns. We then applied our trained model to predict the impact of these putative H3K27ac modifications on gene expression (<xref ref-type="fig" rid="fig1">Figure 1</xref>). We found that our models, which are designed to predict gene expression values, were effective in ranking relative fold-changes among genes in response to the dCas9-p300 system, achieving a Spearman’s rank correlation of ∼0.8. However, their performance in ranking fold-changes within individual genes was less successful when compared to the prediction of gene expression across cell types from their native epigenetic signatures. We offer possible explanations in the discussion section.</p><fig id="fig1" position="float"><label>Figure 1.</label><caption><title>Schematic of the epigenome editing prediction pipeline.</title><p>The pipeline uses epigenetic data to train models to predict endogenous gene expression. These models were used to predict fold-change in gene expression based on perturbed histone PTM input data, and their predictions were validated using CRISPR-Cas9-based epigenome editing data.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-92991-fig1-v1.tif"/></fig></sec><sec id="s2" sec-type="results"><title>Results</title><sec id="s2-1"><title>Histone PTM data are highly predictive of gene expression</title><p>Genome-scale datasets are required to train models to predict gene expression using histone PTMs. Therefore, we obtained histone PTM ChIP-seq and RNA-seq data for 13 different human cell types from ENCODE (<xref ref-type="bibr" rid="bib55">Schreiber et al., 2020a</xref>; <xref ref-type="bibr" rid="bib69">The ENCODE Project Consortium, 2012</xref>; <xref ref-type="table" rid="app1table1">Appendix 1—table 1</xref>). We inspected metagene plots (histone PTMs averaged across genes within gene expression quantiles) describing 6 histone PTMs in each of these 13 different cell types. Based on different overall signal levels across cell types, we concluded that batch effects, likely due to inconsistent sequencing depths, would need to be corrected prior to training models (<xref ref-type="fig" rid="fig2s1">Figure 2—figure supplement 1</xref>).</p><p>We corrected these batch effects by adapting S3norm (<xref ref-type="bibr" rid="bib73">Xiang et al., 2020</xref>; ‘Materials and methods’, <xref ref-type="fig" rid="fig2s2">Figure 2—figure supplement 2</xref>). These corrected histone PTM tracks were then used for the remainder of our analyses along with RNA-seq data for each of the 13 cell types (<xref ref-type="fig" rid="fig2s3">Figure 2—figure supplement 3</xref>).</p><p>Importantly, we observed that H3K27ac and H3K4me3 histone PTM signal strengths positively covaried with gene expression quantile (representative cell types shown in <xref ref-type="fig" rid="fig2">Figure 2</xref>; all cell types shown in <xref ref-type="fig" rid="fig2s3">Figure 2—figure supplement 3</xref>). Conversely, repressive histone PTMs such as H3K27me3 and H3K9me3 were strongly inversely correlated with gene expression quantiles. Spatial patterns in the metagene plots for H3K36me3 suggested that this mark covaried more strongly with gene expression in the gene body than near the TSS. Taken together, these observations recapitulated the current understanding of these well-studied histone PTMs with respect to their associations to gene expression (<xref ref-type="bibr" rid="bib31">Kimura, 2013</xref>; <xref ref-type="bibr" rid="bib44">Millán-Zambrano et al., 2022</xref>; <xref ref-type="bibr" rid="bib77">Zhao et al., 2021</xref>).</p><fig-group><fig id="fig2" position="float"><label>Figure 2.</label><caption><title>Metagene plots show histone post-translational modifications (PTMs) are consistent across cell types and recapitulate established relationships between histone PTMs and gene expression.</title><p>Colors represent genes binned into quantiles based on gene expression. Blue 75–100%, orange 50–75%, green 25–50%, and red 0–25% of gene expression within a cell type. The <inline-formula><alternatives><mml:math id="inf1"><mml:mi>y</mml:mi></mml:math><tex-math id="inft1">\begin{document}$y$\end{document}</tex-math></alternatives></inline-formula>-axis represents <inline-formula><alternatives><mml:math id="inf2"><mml:mo>−</mml:mo><mml:msub><mml:mi>log</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mn>10</mml:mn></mml:mrow></mml:msub></mml:math><tex-math id="inft2">\begin{document}$-\log_{10}$\end{document}</tex-math></alternatives></inline-formula>(<italic>p</italic>-value) obtained from ChIP-seq data.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-92991-fig2-v1.tif"/></fig><fig id="fig2s1" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 1.</label><caption><title>Metagene plots for different cell types for uncorrected ChIP-seq data across gene expression quantiles.</title><p>Blue is the highest and red is the lowest gene expression quantile. ∗ represents data from HEK293 and (A) represents Avocado imputed data.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-92991-fig2-figsupp1-v1.tif"/></fig><fig id="fig2s2" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 2.</label><caption><title>S3norm-based approach for correcting ChIP-seq <inline-formula><alternatives><mml:math id="inf3"><mml:mo>−</mml:mo><mml:msub><mml:mi>log</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mn>10</mml:mn></mml:mrow></mml:msub><mml:mo>⁡</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:mtext>p-values</mml:mtext><mml:mo stretchy="false">)</mml:mo></mml:math><tex-math id="inft3">\begin{document}$-\log_{10}(\text{p-values})$\end{document}</tex-math></alternatives></inline-formula>.</title><p>On the left panel, the <italic>p</italic>-values of a target cell type’s ChIP-seq data, which are to be corrected are plotted on the Y-axis. While the ChIP-seq data for the reference cell type, chosen to be IMR-90, is shown on the X-axis. After correction with this devised procedure, the resulting corrected <italic>p</italic>-values are shown on the Y-axis of the right panel.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-92991-fig2-figsupp2-v1.tif"/></fig><fig id="fig2s3" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 3.</label><caption><title>Metagene plots for different cell types for batch effect-corrected ChIP-seq data across gene expression quantiles.</title><p>Blue is the highest and red is the lowest gene expression quantile. ∗ represents data from HEK293 instead and (A) represents Avocado imputed data.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-92991-fig2-figsupp3-v1.tif"/></fig></fig-group></sec><sec id="s2-2"><title>Histone PTMs accurately predict endogenous gene expression</title><p>To predict how epigenome editing affects gene expression, we first trained models to predict gene expression from endogenous histone PTMs. We trained several convolutional neural networks (CNNs) and ridge regression models to predict the gene expression of each gene in each of the 13 cell types, using only histone PTM data proximal to the TSS as features (‘Materials and methods’, <xref ref-type="fig" rid="fig2">Figure 2</xref>). We observed that Spearman’s rank correlation between the true gene expression and the models’ predicted gene expression on held-out chromosomes improves as the input context size increases; and for all input context sizes, the CNNs outperform ridge regression models (<xref ref-type="fig" rid="fig3">Figure 3A</xref>). Therefore, for the remainder of the analyses, we use a context size of 10,000 base pairs.</p><fig-group><fig id="fig3" position="float"><label>Figure 3.</label><caption><title>Histone post-translational modifications (PTMs) accurately predict endogenous gene expression.</title><p>(<bold>A</bold>) Spearman correlation on genes from held-out chromosomes for different input context lengths, with all cell types pooled together. The blue curve is the mean across 10 computational replicates of convolutional neural networks (CNNs), and the red is the mean across 10 computational replicates of ridge regression. Shaded area represents standard deviation in the Spearman correlation across the 10 computational replicates. (<bold>B</bold>) Spearman correlation on genes of cell types held out during training. The bar plots represent the mean across 10 computational replicates, and the error bars represent the corresponding standard deviations. (<bold>C</bold>) Distribution of Spearman correlations across genes, computed for each gene in test chromosomes by comparing predictions across the 13 cell types. The different curves represent 10 computational replicates for each model type.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-92991-fig3-v1.tif"/></fig><fig id="fig3s1" position="float" specific-use="child-fig"><label>Figure 3—figure supplement 1.</label><caption><title>Spearman correlation distribution across all cell types, for each cell type.</title><p>Each panel corresponds to a different assay where the epigenetic data for that assay in chromosome 17 (which is part of the test dataset) is considered.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-92991-fig3-figsupp1-v1.tif"/></fig><fig id="fig3s2" position="float" specific-use="child-fig"><label>Figure 3—figure supplement 2.</label><caption><title>Endogenous RNA-seq expression levels of HEK293 and HEK293T cell lines are highly concordant.</title><p>Spearman correlations between transcripts per million (TPM) values from RNA-seq datasets of two biological replicates of the HEK293T cell line (with SRA accessions shown in parentheses) are on par with Spearman correlation with RNA-seq TPM values for the HEK293 cell line.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-92991-fig3-figsupp2-v1.tif"/></fig><fig id="fig3s3" position="float" specific-use="child-fig"><label>Figure 3—figure supplement 3.</label><caption><title>Benchmarking of models predicting gene expression from histone marks.</title><p>AUROC values were computed after converting gene expression values into <italic>high</italic> and <italic>low</italic> values using the median gene expression in the training dataset as the threshold.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-92991-fig3-figsupp3-v1.tif"/></fig></fig-group><p>To assess the models’ ability to generalize to unseen cell types, we trained a set of 10 models for each cell type. In particular, we held out the histone PTMs for a given cell type during training and then tested the models on that held-out cell type.</p><p>We observed that the CNNs outperformed ridge regression models on this cross-cell type generalization task across essentially all cell types (<xref ref-type="fig" rid="fig3">Figure 3B</xref>). The reduced performance on the adrenal cell type may be driven by a cell-type-specific biological mechanism that leads to a lower correlation of its epigenetic data with other cell types, particularly for H3K36me3 (<xref ref-type="fig" rid="fig3s1">Figure 3—figure supplement 1</xref>).</p><p>Although our models accurately predicted endogenous gene expression, this does not guarantee their ability to accurately predict the relationship between local histone PTM variations and gene expression for a particular gene across different cell types. Therefore, we determined Spearman’s rank correlations between the observed expression and the predicted expression for each held-out gene across the different cell types. The distribution of these correlations suggests that overall the CNNs can better rank cell types by gene expression than ridge regression (<xref ref-type="fig" rid="fig3">Figure 3C</xref>). In particular, the median cross-cell type correlation is ∼0.53 for CNNs compared to ∼0.39 for ridge regression.</p><p>We also benchmarked the predictive performance of our CNN model against existing methods and observed that we outperform all existing methods across multiple cell types (<xref ref-type="fig" rid="fig3s3">Figure 3—figure supplement 3</xref>).</p></sec><sec id="s2-3"><title>Models recover established relationships between histone PTMs and gene expression</title><p>We investigated what features of the data the models used to predict gene expression. For a given gene, we modified the input histone PTMs one-by-one at nucleosome-scale and measured the predicted fold-change in gene expression (<xref ref-type="fig" rid="fig4">Figure 4</xref>, <xref ref-type="fig" rid="fig4s1">Figure 4—figure supplement 1</xref>, ‘Materials and methods’).</p><fig-group><fig id="fig4" position="float"><label>Figure 4.</label><caption><title>Features learned by gene expression models.</title><p>Each point on the X-axis corresponds to in silico perturbation of that assay at that position, and the Y-axis measures the predicted fold-change in gene expression, averaged across a set of 100 trained models. The fold-changes were averaged across 500 randomly chosen genes.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-92991-fig4-v1.tif"/></fig><fig id="fig4s1" position="float" specific-use="child-fig"><label>Figure 4—figure supplement 1.</label><caption><title>Features learned by gene expression models for H3K9me3 in K562.</title><p>Each point on the X-axis corresponds to in silico perturbation of H3K9me3 at that position and the Y-axis measures the predicted fold-change in gene expression, averaged across a set of 100 trained models. The fold-changes were averaged across 500 randomly chosen genes. This is a zoomed-in version of the subplot in <xref ref-type="fig" rid="fig4">Figure 4</xref> corresponding to H3K9me3 in K562.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-92991-fig4-figsupp1-v1.tif"/></fig></fig-group><p>We observed considerable changes to the predicted fold-change upon modifying different histone PTMs. In particular, our CNN models predict that repressive marks such as H3K27me3 and H3K9me3 proximal to the TSS result in a slight decrease in expression. In contrast, activating histone PTMs such as H3K27ac and H3K4me3 results in an almost twofold increase in predicted gene expression near the TSS. Activating both of these markers exhibits a periodic pattern, likely reflecting nucleosome occupancy. However, activation of H3K4me3 results in a sharp increase in gene expression downstream of the TSS. Additionally, we observed that H3K36me3 is predicted to increase expression, but only if it is deposited in the gene body, and the degree of activation gradually increases as it is deposited further inside of the gene body. The consistency of these observations with established mechanisms, observed previously in the literature, via which these histone PTMs modulate gene expression (<xref ref-type="bibr" rid="bib31">Kimura, 2013</xref>) lends credence to our gene expression models and shows that these models learn the spatial patterns of histone PTMs.</p></sec><sec id="s2-4"><title>dCas9-p300 differentially activates genes depending on gRNA-targeted site</title><p>To test if our gene expression models could accurately predict the outcome of in situ epigenome editing experiments, we first generated dCas9-p300 data in the HEK293T cell line for eight genes (<xref ref-type="fig" rid="fig5">Figure 5</xref>). We assayed at least five gRNAs per gene with at least three replicates for each gRNA. We used the HEK293T cell line because it is a widely used testbed for epigenome editing strategies (<xref ref-type="bibr" rid="bib23">Hilton et al., 2015</xref>; <xref ref-type="bibr" rid="bib46">Nuñez et al., 2021</xref>; <xref ref-type="bibr" rid="bib47">O’Geen et al., 2017</xref>; <xref ref-type="bibr" rid="bib39">Mahata et al., 2023</xref>; <xref ref-type="bibr" rid="bib11">Escobar et al., 2022</xref>; <xref ref-type="bibr" rid="bib71">Wang et al., 2022</xref>). Based on <xref ref-type="fig" rid="fig2">Figures 2</xref> and <xref ref-type="fig" rid="fig4">4</xref>, the largest changes in H3K27ac across gene expression quantiles occur within 500 base pairs of the TSS, so we constrained gRNA targeting to this critical window. We filtered gRNAs for predicted specificity (<xref ref-type="bibr" rid="bib7">Concordet and Haeussler, 2018</xref>) and on-target activity scores (<xref ref-type="bibr" rid="bib51">Sanson et al., 2018</xref>). Each gRNA was tested individually, and relative mRNA abundance was measured using quantitative PCR (qPCR).</p><fig-group><fig id="fig5" position="float"><label>Figure 5.</label><caption><title>dCas9-p300 epigenome editing at eight endogenous genes identifies gene-specific responses.</title><p>The genes tested are <italic>CYP17A1</italic>, <italic>SOX11</italic>, <italic>C2CD4B</italic>, <italic>CXCR4</italic>, <italic>CD79A</italic>, <italic>TGFBR1</italic>, <italic>MYO1G</italic>, and <italic>PRSS12</italic>. (<bold>A</bold>) gRNA (n=5) targeting <inline-formula><alternatives><mml:math id="inf4"><mml:mo>+</mml:mo><mml:mrow class="MJX-TeXAtom-ORD"><mml:mo>/</mml:mo></mml:mrow><mml:mo>−</mml:mo></mml:math><tex-math id="inft4">\begin{document}$+/-$\end{document}</tex-math></alternatives></inline-formula> 250 bp of each gene was selected. (<bold>B</bold>) These selected gRNA were individually co-transfected with dCas9-p300 with relative mRNA determined with qPCR. (<bold>C</bold>) Relative mRNA associated with selected guide position is displayed with the highest activating guide position marked in orange. The Y-axis corresponds to qPCR fold-change.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-92991-fig5-v1.tif"/></fig><fig id="fig5s1" position="float" specific-use="child-fig"><label>Figure 5—figure supplement 1.</label><caption><title>Transfection efficiency is shared across experiments.</title><p>This figure shows consistent transfection efficiency across multiple gene targets. Histograms show the distribution of fluorescent signal intensity, indicating the percentage of cells (right) successfully transfected with the reporter construct containing mCherry-p300. We selected two gRNAs (gRNA1 and gRNA2) for two gene targets (<italic>CYP17A1</italic> and <italic>SOX11</italic>) and a scramble gRNA to measure the transfection efficiency. An average transfection efficiency of 17% was achieved across the different samples with no transfection in the untreated cells.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-92991-fig5-figsupp1-v1.tif"/></fig></fig-group><p>We successfully increased gene expression of all eight genes with fold-change activation using the most effective respective gRNA for each gene ranging from 3-fold to ∼6500-fold relative to a non-targeting control gRNA (<xref ref-type="fig" rid="fig5">Figure 5C</xref>). Some of this variation may be explained by differences in endogenous gene expression levels, with the targeting of lowly expressed genes resulting in higher fold-change measurements (<xref ref-type="table" rid="app1table2">Appendix 1—table 2</xref>), as observed previously (<xref ref-type="bibr" rid="bib71">Wang et al., 2022</xref>). Nevertheless, substantial variability was observed in gRNA efficacy for all targeted genes. In particular, two (<italic>MYO1G</italic> and <italic>PRSS12</italic>) out of eight genes had the most efficacious gRNA downstream of the TSS. This contrasts with other reports where targeting CRISPR/Cas-based activators upstream of the TSS leads to the highest activation (<xref ref-type="bibr" rid="bib45">Mohr et al., 2016</xref>; <xref ref-type="bibr" rid="bib18">Gilbert et al., 2014</xref>).</p><p>These data indicate that the rules governing the outcomes for successful dCas9-p300-based epigenome editing – and subsequent increased transcriptional activation – are complex and highlight the fact that locus-specific nuances can be important factors in epigenome editing experiments. For example, two gRNAs targeting within ∼50 base pairs of each other on <italic>C2CD4B</italic> have a 100-fold difference in measured mRNA (<xref ref-type="fig" rid="fig5">Figure 5C</xref>). Further, gRNAs targeting the same position in different genes can have vastly different effects. For instance, several gRNAs targeting ∼250 base pairs upstream of the <italic>CYP17A1</italic> TSS result in a high fold-change, while two gRNAs targeting roughly the same position in <italic>MYO1G</italic> failed to produce substantial activation (<xref ref-type="fig" rid="fig5">Figure 5C</xref>).</p></sec><sec id="s2-5"><title>Computationally predicting the outcome of dCas9-p300 epigenome editing experiments</title><p>To test the hypothesis that dCas9-p300 acts through the local deposition of H3K27ac, we modeled this process in silico and used these perturbations as inputs to our models trained on endogenous gene expression.</p><p>We modeled the effect of dCas9-p300 on histone PTMs based on evidence from the literature as well as additional experiments we performed. The key assumptions of this model are (1) there exists steric hindrance of dCas9 by nucleosomes (<xref ref-type="bibr" rid="bib40">Makasheva et al., 2021</xref>; <xref ref-type="bibr" rid="bib25">Horlbeck et al., 2016</xref>; <xref ref-type="bibr" rid="bib26">Isaac et al., 2016</xref>; <xref ref-type="bibr" rid="bib48">Radzisheuskaya et al., 2016</xref>); (2) dCas9-p300 acts locally, altering H3K27ac levels near the gRNA target locus (<xref ref-type="bibr" rid="bib16">Gemberling et al., 2021</xref>; <xref ref-type="bibr" rid="bib10">Dominguez et al., 2022</xref>) (we adopted this simplifying assumption since off-target effects are unpredictable and underexplored; <xref ref-type="bibr" rid="bib10">Dominguez et al., 2022</xref>; <xref ref-type="bibr" rid="bib16">Gemberling et al., 2021</xref>; <xref ref-type="bibr" rid="bib72">Weinert et al., 2018</xref>); and (3) dCas9-p300 can deposit H3K27ac at nucleosomes, as defined by MNase activity (see ‘Materials and methods’; <xref ref-type="bibr" rid="bib57">Segelle et al., 2022</xref>; <xref ref-type="bibr" rid="bib80">Zhou et al., 2016</xref>). Our resulting in silico perturbation model had a number of free parameters that we briefly describe below. Wherever possible, we used values for these parameters obtained from the literature or tested a range of plausible values. For a more complete description of the model, see ‘Materials and methods’.</p><p>The first component of our perturbation model is steric hindrance of dCas9-p300 by nucleosomes (<xref ref-type="fig" rid="fig6">Figure 6A</xref>). Intuitively, if DNA is tightly wound around a nucleosome, the gRNA would be less likely to bind successfully. Mathematically, we modeled this as an inverse relationship between the amount of H3K27ac deposited and the MNase activity at the gRNA target locus.</p><fig-group><fig id="fig6" position="float"><label>Figure 6.</label><caption><title>In silico model for dCas9-p300-based epigenome editing.</title><p>(<bold>A</bold>) dCas9-p300 is more likely to bind to a position not occupied by the nucleosome. Thicker green arrow represents higher probability of binding for a gRNA targeting that site. (<bold>B</bold>) The in silico perturbation is modeled as a Gaussian kernel parameterized by a standard deviation, <inline-formula><alternatives><mml:math id="inf5"><mml:mi>σ</mml:mi></mml:math><tex-math id="inft5">\begin{document}$\sigma$\end{document}</tex-math></alternatives></inline-formula>, and the amount of H3K27ac deposited, <italic>λ</italic>. (<bold>C</bold>) The final perturbed H3K27ac is obtained by point-wise multiplication of the Gaussian kernel with nucleosome occupancy quantified by MNase activity since dCas9-p300 can only acetylate histones within nucleosomes. (<bold>D</bold>) Ranks for predicted and endogenous expression across 8 genes and 13 cell types. Rank 1 corresponds to the highest numerical value. (<bold>E</bold>) Ranks for predicted and empirically measured expression fold-changes following perturbation by dCas9-p300 for eight genes in HEK293T cells. Rank 1 corresponds to the highest numerical value.</p><p><supplementary-material id="fig6sdata1"><label>Figure 6—source data 1.</label><caption><title>Raw qPCR data.</title><p>Each row has an individual measurement which includes pertinent information used to generate in silico and compare with model predictions. Columns include corresponding guide information regarding gRNA position and coordinates as well as gene information such as orientation and coordinates.</p></caption><media mimetype="application" mime-subtype="xlsx" xlink:href="elife-92991-fig6-data1-v1.xlsx"/></supplementary-material></p><p><supplementary-material id="fig6sdata2"><label>Figure 6—source data 2.</label><caption><title>Raw CUT&amp;RUN qPCR data.</title><p>This table includes measurements with corresponding sgRNA used and their distance with respect to the TSS. Gene information and amplicon centerpoint distance to the TSS.</p></caption><media mimetype="application" mime-subtype="xlsx" xlink:href="elife-92991-fig6-data2-v1.xlsx"/></supplementary-material></p><p><supplementary-material id="fig6sdata3"><label>Figure 6—source data 3.</label><caption><title>Primer sequences, sources, assay use, and corresponding direction.</title><p>CUT&amp;RUN primers have their corresponding genomic coordinates reported corresponding to the regions they amplify.</p></caption><media mimetype="application" mime-subtype="xlsx" xlink:href="elife-92991-fig6-data3-v1.xlsx"/></supplementary-material></p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-92991-fig6-v1.tif"/></fig><fig id="fig6s1" position="float" specific-use="child-fig"><label>Figure 6—figure supplement 1.</label><caption><title>H3K27ac levels elevation is similar across quantified regions following gRNA dCas9-p300 targeting.</title><p>Each colored line corresponds to a gRNA targeting proximal to CXCR4 and TGFBR1 in HEK293T cells. The X-axis represents the distance between gRNA and the CUT&amp;RUN amplicon. The Y-axis represents H3K27ac fold enrichment estimated through CUT&amp;RUN.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-92991-fig6-figsupp1-v1.tif"/></fig><fig id="fig6s2" position="float" specific-use="child-fig"><label>Figure 6—figure supplement 2.</label><caption><title>Gene-wise predicted vs. experimental gene expression transcripts per million (TPM) ranks.</title><p>Each dot corresponds to a cell type, and the title of each plot shows the Spearman correlation and the corresponding <italic>p</italic>-values. Rank 1 corresponds to the highest numerical value.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-92991-fig6-figsupp2-v1.tif"/></fig><fig id="fig6s3" position="float" specific-use="child-fig"><label>Figure 6—figure supplement 3.</label><caption><title>Gene-wise predicted vs experimental fold-change ranks.</title><p>Each dot corresponds to a gRNA targeting a locus near the TSS of the gene (each gRNA corresponds to at least three replicates and hence the fold-change shown here is the experimental mean). Rank 1 corresponds to the highest numerical value.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-92991-fig6-figsupp3-v1.tif"/></fig><fig id="fig6s4" position="float" specific-use="child-fig"><label>Figure 6—figure supplement 4.</label><caption><title>Predicted vs. experimental fold-change ranks.</title><p>Each dot corresponds to a gRNA targeting a locus near the TSS of the gene. Rank 1 corresponds to the highest numerical value.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-92991-fig6-figsupp4-v1.tif"/></fig></fig-group><p>It is widely assumed that dCas9-p300 activates genes through the local deposition of H3K27ac (<xref ref-type="bibr" rid="bib32">Klann et al., 2017</xref>; <xref ref-type="bibr" rid="bib10">Dominguez et al., 2022</xref>). To model this, we increased local levels of H3K27ac relative to endogenous levels according to a Gaussian kernel centered at the gRNA target locus (<xref ref-type="fig" rid="fig6">Figure 6B</xref>). This adds acetylation primarily within a distance controlled by the standard deviation (<inline-formula><alternatives><mml:math id="inf6"><mml:mi>σ</mml:mi></mml:math><tex-math id="inft6">\begin{document}$\sigma$\end{document}</tex-math></alternatives></inline-formula>) of the kernel. We performed CUT&amp;RUN experiments (see Appendix 1) that suggest that this distance is at least 1000 base pairs (<xref ref-type="fig" rid="fig6s1">Figure 6—figure supplement 1</xref>). Since we also do not know the degree to which dCas9-p300 alters H3K27ac levels, we modeled this as another free parameter, <italic>λ</italic>, which we varied over a range of plausible values (‘Materials and methods’).</p><p>Finally, we assumed that dCas9-p300 does not affect the positioning of nucleosomes and hence can only add H3K27ac at positions currently occupied by histones (<xref ref-type="bibr" rid="bib80">Zhou et al., 2016</xref>). As such, we expect H3K27ac levels to only increase at loci where there is MNase activity. In particular, we modulated the Gaussian kernel described above by performing point-wise multiplication with MNase activity (<xref ref-type="fig" rid="fig6">Figure 6C</xref>).</p><p>Since nucleosome positioning plays a crucial role in our perturbation model, we generated, to our knowledge, the first MNase-seq data for the HEK293T cell line (see Appendix 1).</p><p>To get a baseline of how well our perturbation model might be able to predict the effect of dCas9-p300 on gene expression, we considered the 13 distinct cell types as being analogous to natural perturbations of local histone PTMs. Across the eight genes discussed above, which were excluded from the training set, we observed a Spearman’s rank correlation of ∼0.8 between the endogenous expression and that predicted by our expression model (<xref ref-type="fig" rid="fig6">Figure 6D</xref>). This correlation was in line with the correlation observed across the endogenous transcriptome (<xref ref-type="fig" rid="fig3">Figure 3A and B</xref>). We further observed that our expression models were able to accurately rank gene expression across cell types within individual genes (<xref ref-type="fig" rid="fig6s2">Figure 6—figure supplement 2</xref>).</p><p>We then computed fold-changes between the expression predicted using endogenous histone PTMs and the expression predicted using in silico perturbations of these histone PTMs. We observed that our models were effective in ranking relative fold-changes across genes in response to dCas9-p300, achieving a Spearman’s rank correlation of ∼0.8 between these predicted fold-changes and the experimentally determined mRNA fold-changes induced by dCas9-p300 (<xref ref-type="fig" rid="fig6">Figure 6E</xref>). However, the performance in ranking fold-changes within individual genes was less accurate (<xref ref-type="fig" rid="fig6s3">Figure 6—figure supplement 3</xref>) when compared to the prediction of cell-type-specific gene expression from native epigenetic signatures (<xref ref-type="fig" rid="fig6s2">Figure 6—figure supplement 2</xref>).</p><p>We extended this analysis to a Perturb-seq dataset, consisting of gRNAs targeting proximal to the TSS of five genes in the K562 cell line to further assess the model’s ability to estimate gene expression changes. Consistent with the performance observed in <xref ref-type="fig" rid="fig6">Figure 6E</xref>, the model demonstrated robustness in predicting gene expression fold-changes across these 27 gRNAs targeting these five genes. Notably, these predictions achieved a Spearman’s rank correlation of ∼ 0.47 with the experimentally determined mRNA fold-changes measured by Perturb-seq, as shown in <xref ref-type="fig" rid="fig6s4">Figure 6—figure supplement 4</xref> (see Appendix 1). These results reinforce the model’s effectiveness in capturing the nuanced effects of epigenome editing across different genes and cell types.</p></sec></sec><sec id="s3" sec-type="discussion"><title>Discussion</title><p>Here, we sought to investigate whether we could predict how targeted epigenome editing affects endogenous gene expression. First, we collected data from ENCODE which reflects how PTMs to histones covary with gene expression across cell types. We trained models to predict endogenous gene expression from these histone PTMs and found that these models were highly predictive (<xref ref-type="fig" rid="fig3">Figure 3</xref>). We further showed that such models learned known relationships between histone PTMs and gene expression (<xref ref-type="fig" rid="fig4">Figure 4</xref>). To test whether these expression models could predict the outcomes of epigenome editing experiments, we generated dCas9-p300 epigenome editing data in the HEK293T cell line for eight genes along with genome-wide MNase-seq data for this testbed cell line. We anticipate that the genome-wide nucleosome occupancy information for the HEK293T cell line provided by our MNase-seq experiment will be a useful resource for the genomics community. We also generated dCas9-p300 epigenome editing data via a Perturb-seq experiment in K562 cells with gRNAs targeting the promoter regions of five genes. In this study, we focused on the histone changes induced by dCas9-p300 epigenome editing, but future studies may use the framework described in our manuscript and apply it to other transcriptional editors as well.</p><p>We modeled dCas9-p300’s impact on local H3K27ac using a variety of parameter choices and found that these models accurately predicted fold-changes across genes. However, they were less accurate at predicting the outcome of these experiments within a given gene, as compared to predicting gene expression from the endogenous epigenetic signatures (<xref ref-type="fig" rid="fig6s2">Figure 6—figure supplements 2 and</xref> <xref ref-type="fig" rid="fig6s3">3</xref>). Since the endogenous epigenetic signatures could be different across genes, these <italic>global</italic> factors might drive the models’ accurate inter-gene fold-change prediction accuracy. However, since ranking fold-changes within a gene requires a detailed understanding of the epigenetic profiles before and after dCas9-p300 epigenome editing, the reduction in performance from predicting endogenous expression to predicting the outcome of epigenome editing experiments is likely explained by one or more of the following hypotheses: (1) dCas9-p300 activates gene expression by mechanisms other than the <italic>local</italic> acetylation of H3K27 or dCas9-p300 functions differently from native p300; (2) differences in gRNA efficacy are not accurately explained by existing computational scores; or (3) our models, trained on endogenous gene expression across various cell types, failed to generalize even if dCas9-p300 perturbations are correctly modeled. We discuss these possible explanations more in depth below.</p><p>We considered numerous models of how dCas9-p300 affects local histone PTMs. These models span current hypotheses of how dCas9-p300 alters local histone PTMs such as H3K27ac. The poor generalization of our models in predicting intra-gene epigenome editing fold-changes could be explained by dCas9-p300 acting via mechanisms beyond local acetylation of histone proteins and H3K27 (<xref ref-type="bibr" rid="bib77">Zhao et al., 2021</xref>). For example, p300 is a promiscuous lysine acetyltransferase and dCas9-p300 could be broadly acetylating across the proteome impacting <italic>trans</italic> factors (<xref ref-type="bibr" rid="bib72">Weinert et al., 2018</xref>). Alternatively, local acetylation could be contingent on unmodeled factors such as <italic>trans</italic>-acting proteins or other histone PTMs present at the locus (<xref ref-type="bibr" rid="bib77">Zhao et al., 2021</xref>; <xref ref-type="bibr" rid="bib78">Zheng et al., 2021</xref>). Furthermore, the genome-wide specificity of dCas9-p300-mediated histone acetylation – although likely better than small molecule-based perturbations – remains imperfect (<xref ref-type="bibr" rid="bib16">Gemberling et al., 2021</xref>; <xref ref-type="bibr" rid="bib10">Dominguez et al., 2022</xref>). Our inability to accurately predict the relative fold-change of different gRNAs targeting the same gene suggests that these unmodeled factors would have to differentially affect neighboring loci within the same gene. This highlights that the current understanding of the mechanism via which dCas9-p300 drives gene expression is potentially incomplete. To better understand this mechanism, it would be immensely helpful to generate a compendium of histone PTM profiles before and after performing epigenome editing, which would enable us to train better machine learning models to predict the impact of dCas9-p300 on gene expression.</p><p>Another possible explanation for the drop in accuracy is varying gRNA efficacies. For example, gRNAs might have different levels of on-target and off-target effects. Although we ensured that all of the gRNAs used in generating the dCas9-p300 epigenome editing data were predicted to have high on-target and low off-target scores, we observed examples of gRNAs that targeted roughly the same genomic position but had vastly different impacts on gene expression. This suggests that these differences could be driven by inconsistencies in gRNA efficacy instead of local acetylation dynamics. Generating a large number of pairs of gRNAs, such as through CRISPR screens (<xref ref-type="bibr" rid="bib54">Schmidt et al., 2022</xref>), targeting nearby positions could help to elucidate the factors that drive differential gRNA efficacy for epigenome editing.</p><p>The ambiguity in how to accurately model the impact of epigenome editing stands in contrast to the simpler case of DNA sequence changes, where perturbations are relatively trivial to model. Indeed, dCas9-p300 changes histone PTMs in complex ways, rendering the modeling of such perturbations much more challenging. In contrast, models like Enformer (<xref ref-type="bibr" rid="bib1">Avsec et al., 2021</xref>) that predict gene expression directly from DNA sequence may be able to generalize to DNA sequence perturbations better due to their relative simplicity.</p><p>Another source of generalization error could be extrapolating beyond the range of the training data. Massively increasing the amount of H3K27ac at a locus may make a gene look different than any other endogenous gene observed during training. Regression approaches including neural networks are known to have limitations in extrapolation (<xref ref-type="bibr" rid="bib74">Xu et al., 2020</xref>).</p><p>Our research indicates that we can predict endogenous gene expression accurately based on histone PTMs. By creating a comprehensive dataset of epigenome editing, which assays histone PTMs before and after in situ perturbations, we can enhance machine learning models. This will improve our understanding of the effects of dCas9-p300 on gene expression and assist in the design of gRNAs for achieving fine-tuned control over gene expression levels. These advancements are vital for devising experiments that deepen our mechanistic insight and offer effective strategies for human epigenome editing.</p></sec><sec id="s4" sec-type="materials|methods"><title>Materials and methods</title><sec id="s4-1"><title>Data preparation</title><p>We obtained <inline-formula><alternatives><mml:math id="inf7"><mml:mstyle><mml:mrow><mml:mstyle displaystyle="false"><mml:mo>−</mml:mo><mml:msub><mml:mi>log</mml:mi><mml:mrow><mml:mn>10</mml:mn></mml:mrow></mml:msub><mml:mo>⁡</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>p</mml:mi><mml:mtext>-value</mml:mtext></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mstyle></mml:mrow></mml:mstyle></mml:math><tex-math id="inft7">\begin{document}$-\log_{10}({p\text{-value}})$\end{document}</tex-math></alternatives></inline-formula> ChIP-seq tracks created by running the MACS2 peak-caller (<xref ref-type="bibr" rid="bib13">Feng et al., 2012</xref>) on read count data, from the ENCODE Imputation Challenge (<xref ref-type="bibr" rid="bib55">Schreiber et al., 2020a</xref>). For three tracks where data were not available, we downloaded Avocado (<xref ref-type="bibr" rid="bib56">Schreiber et al., 2020b</xref>) imputations from the ENCODE data portal (<xref ref-type="bibr" rid="bib69">The ENCODE Project Consortium, 2012</xref>). We binned each epigenetic track at 25 base pair resolution and pre-processed them with an additional <inline-formula><alternatives><mml:math id="inf8"><mml:mi>log</mml:mi></mml:math><tex-math id="inft8">\begin{document}$\log$\end{document}</tex-math></alternatives></inline-formula> operation before inputting them into the models for training.</p><p>We downloaded polyA-plus RNA-seq gene expression transcripts per million (TPM) values for each of the 13 cell types in <xref ref-type="table" rid="app1table1">Appendix 1—table 1</xref>, from the ENCODE data portal (<xref ref-type="bibr" rid="bib69">The ENCODE Project Consortium, 2012</xref>) and preprocessed them with a <inline-formula><alternatives><mml:math id="inf9"><mml:mi>log</mml:mi></mml:math><tex-math id="inft9">\begin{document}$\log$\end{document}</tex-math></alternatives></inline-formula> operation.</p></sec><sec id="s4-2"><title>Normalizing <italic>p</italic>-values by adapting S3norm</title><p>We assigned IMR-90 to be a reference cell type, for each of the six histone PTMs and kept its <italic>p</italic>-values unchanged. We then performed a transformation for each of the remaining cell types adapted from the core technique developed by S3norm (<xref ref-type="bibr" rid="bib73">Xiang et al., 2020</xref>), in order to normalize each histone PTM track in each of these remaining cell types, with respect to the corresponding histone PTM track in IMR-90.</p><p>First, we computed <italic>peaks</italic> in both the reference as well as the target cell type. <italic>Peaks</italic> were defined as the 25 base pair bins corresponding to FDR-adjusted <italic>p</italic>-values &lt;0.05 (<xref ref-type="bibr" rid="bib4">Benjamini and Hochberg, 1995</xref>). For histone PTM tracks that were obtained from Avocado imputations (due to lack of availability of experimental data), <italic>peaks</italic> were defined to be the 1000 bins containing the smallest Avocado imputed <inline-formula><alternatives><mml:math id="inf10"><mml:mi>p</mml:mi></mml:math><tex-math id="inft10">\begin{document}$p$\end{document}</tex-math></alternatives></inline-formula>-values (based on suggestions from the authors of Avocado <xref ref-type="bibr" rid="bib56">Schreiber et al., 2020b</xref>). All the remaining bins were defined to be <italic>background</italic>, for both the reference as well as the target cell types.</p><p>We then computed the list of <italic>peaks</italic> that were common to both the reference and the target cell types. These were termed <italic>common peaks</italic>. Similarly, we defined <italic>common background</italic> as the list of bins that were assigned to be <italic>background</italic> in both the reference as well as the target cell types.</p><p>The S3norm method was designed to work with count data, which is always ≥ 1. However, the histone PTM tracks, which are represented as <inline-formula><alternatives><mml:math id="inf11"><mml:mstyle><mml:mrow><mml:mstyle displaystyle="false"><mml:mo>−</mml:mo><mml:msub><mml:mi>log</mml:mi><mml:mrow><mml:mn>10</mml:mn></mml:mrow></mml:msub><mml:mo>⁡</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>p</mml:mi><mml:mtext>-values</mml:mtext></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mstyle></mml:mrow></mml:mstyle></mml:math><tex-math id="inft11">\begin{document}$-\log_{10}({p\text{-values}})$\end{document}</tex-math></alternatives></inline-formula>, are not guaranteed to always be ≥ 1; hence, we transformed all the histone PTM tracks by adding 1 to the <inline-formula><alternatives><mml:math id="inf12"><mml:mstyle><mml:mrow><mml:mstyle displaystyle="false"><mml:mo>−</mml:mo><mml:msub><mml:mi>log</mml:mi><mml:mrow><mml:mn>10</mml:mn></mml:mrow></mml:msub><mml:mo>⁡</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>p</mml:mi><mml:mtext>-values</mml:mtext></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mstyle></mml:mrow></mml:mstyle></mml:math><tex-math id="inft12">\begin{document}$-\log_{10}({p\text{-values}})$\end{document}</tex-math></alternatives></inline-formula>, in both the reference as well as the target cell types.</p><p>Additionally, since the histone PTM tracks obtained from imputations performed by Avocado were not guaranteed to be distributed similar to experimental <inline-formula><alternatives><mml:math id="inf13"><mml:mstyle><mml:mrow><mml:mstyle displaystyle="false"><mml:mo>−</mml:mo><mml:msub><mml:mi>log</mml:mi><mml:mrow><mml:mn>10</mml:mn></mml:mrow></mml:msub><mml:mo>⁡</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>p</mml:mi><mml:mtext>-values</mml:mtext></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mstyle></mml:mrow></mml:mstyle></mml:math><tex-math id="inft13">\begin{document}$-\log_{10}({p\text{-values}})$\end{document}</tex-math></alternatives></inline-formula>, we scaled all the histone PTM tracks (both experimental as well as Avocado imputations) by dividing them by the minimum observed value in <italic>common peaks</italic> and <italic>common background</italic>, in order to bring experimental data and Avocado imputations onto a similar footing. In particular, before applying the S3norm normalization, we transformed <inline-formula><alternatives><mml:math id="inf14"><mml:mstyle><mml:mrow><mml:mstyle displaystyle="false"><mml:mo>−</mml:mo><mml:msub><mml:mi>log</mml:mi><mml:mrow><mml:mn>10</mml:mn></mml:mrow></mml:msub><mml:mo>⁡</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>p</mml:mi><mml:mtext>-values</mml:mtext></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mstyle></mml:mrow></mml:mstyle></mml:math><tex-math id="inft14">\begin{document}$-\log_{10}({p\text{-values}})$\end{document}</tex-math></alternatives></inline-formula> in <italic>common peaks</italic> and <italic>common background</italic> for both the reference as well as the target cell type as follows:<disp-formula id="equ1"><label>(1)</label><alternatives><mml:math id="m1"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mtable rowspacing="3pt" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true"><mml:mtr><mml:mtd><mml:msub><mml:mtext>TransformedCommonPeaks</mml:mtext><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mtext>reference</mml:mtext></mml:mrow></mml:msub></mml:mtd><mml:mtd><mml:mi/><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn><mml:mo>+</mml:mo><mml:msub><mml:mtext>CommonPeaks</mml:mtext><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mtext>reference</mml:mtext></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:munder><mml:mo movablelimits="true" form="prefix">min</mml:mo><mml:mi>i</mml:mi></mml:munder><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mtext>CommonPeaks</mml:mtext><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mtext>reference</mml:mtext></mml:mrow></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mfrac></mml:mtd></mml:mtr></mml:mtable></mml:mstyle></mml:math><tex-math id="t1">\begin{document}$$\displaystyle \newcommand{\fref}[1]{Figure~\ref{#1}}\newcommand{\sref}[1]{Section~\ref{#1}} \text{TransformedCommonPeaks}_{i,\text{reference}} &amp;= \frac{1 + \text{CommonPeaks}_{i,\text{reference}}}{\min_i(\text{CommonPeaks}_{i,\text{reference}})}$$\end{document}</tex-math></alternatives></disp-formula><disp-formula id="equ2"><label>(2)</label><alternatives><mml:math id="m2"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mtable rowspacing="3pt" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true"><mml:mtr><mml:mtd><mml:msub><mml:mtext>TransformedCommonPeaks</mml:mtext><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mtext>target</mml:mtext></mml:mrow></mml:msub></mml:mtd><mml:mtd><mml:mi/><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn><mml:mo>+</mml:mo><mml:msub><mml:mtext>CommonPeaks</mml:mtext><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mtext>target</mml:mtext></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:munder><mml:mo movablelimits="true" form="prefix">min</mml:mo><mml:mi>i</mml:mi></mml:munder><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mtext>CommonPeaks</mml:mtext><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mtext>target</mml:mtext></mml:mrow></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mfrac></mml:mtd></mml:mtr></mml:mtable></mml:mstyle></mml:math><tex-math id="t2">\begin{document}$$\displaystyle \newcommand{\fref}[1]{Figure~\ref{#1}}\newcommand{\sref}[1]{Section~\ref{#1}} \text{TransformedCommonPeaks}_{i,\text{target}} &amp;= \frac{1 + \text{CommonPeaks}_{i,\text{target}}}{\min_i(\text{CommonPeaks}_{i,\text{target}})}$$\end{document}</tex-math></alternatives></disp-formula><disp-formula id="equ3"><label>(3)</label><alternatives><mml:math id="m3"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mtable rowspacing="3pt" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true"><mml:mtr><mml:mtd><mml:msub><mml:mtext>TransformedCommonBackground</mml:mtext><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mtext>reference</mml:mtext></mml:mrow></mml:msub></mml:mtd><mml:mtd><mml:mi/><mml:mo>=</mml:mo><mml:mtext>max</mml:mtext><mml:mo stretchy="false">(</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn><mml:mo>+</mml:mo><mml:msub><mml:mtext>CommonBackground</mml:mtext><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mtext>reference</mml:mtext></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:munder><mml:mo movablelimits="true" form="prefix">min</mml:mo><mml:mi>i</mml:mi></mml:munder><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mtext>CommonBackground</mml:mtext><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mtext>reference</mml:mtext></mml:mrow></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mfrac><mml:mo>,</mml:mo><mml:mn>0</mml:mn><mml:mo stretchy="false">)</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:mstyle></mml:math><tex-math id="t3">\begin{document}$$\displaystyle \newcommand{\fref}[1]{Figure~\ref{#1}}\newcommand{\sref}[1]{Section~\ref{#1}} \text{TransformedCommonBackground}_{i,\text{reference}} &amp;= \text{max}(\frac{1 + \text{CommonBackground}_{i,\text{reference}}}{\min_i(\text{CommonBackground}_{i,\text{reference}})}, 0)$$\end{document}</tex-math></alternatives></disp-formula><disp-formula id="equ4"><label>(4)</label><alternatives><mml:math id="m4"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mtable rowspacing="3pt" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true"><mml:mtr><mml:mtd><mml:msub><mml:mtext>TransformedCommonBackground</mml:mtext><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mtext>target</mml:mtext></mml:mrow></mml:msub></mml:mtd><mml:mtd><mml:mi/><mml:mo>=</mml:mo><mml:mtext>max</mml:mtext><mml:mo stretchy="false">(</mml:mo><mml:mfrac><mml:mrow><mml:mn>1</mml:mn><mml:mo>+</mml:mo><mml:msub><mml:mtext>CommonBackground</mml:mtext><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mtext>target</mml:mtext></mml:mrow></mml:msub></mml:mrow><mml:mrow><mml:munder><mml:mo movablelimits="true" form="prefix">min</mml:mo><mml:mi>i</mml:mi></mml:munder><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mtext>CommonBackground</mml:mtext><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mtext>target</mml:mtext></mml:mrow></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:mrow></mml:mfrac><mml:mo>,</mml:mo><mml:mn>0</mml:mn><mml:mo stretchy="false">)</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:mstyle></mml:math><tex-math id="t4">\begin{document}$$\displaystyle \newcommand{\fref}[1]{Figure~\ref{#1}}\newcommand{\sref}[1]{Section~\ref{#1}} \text{TransformedCommonBackground}_{i,\text{target}} &amp;= \text{max}(\frac{1 + \text{CommonBackground}_{i,\text{target}}}{\min_i(\text{CommonBackground}_{i,\text{target}})}, 0)$$\end{document}</tex-math></alternatives></disp-formula></p><p>The normalization procedure of S3norm then wishes to find two positive parameters, <inline-formula><alternatives><mml:math id="inf15"><mml:mi>α</mml:mi></mml:math><tex-math id="inft15">\begin{document}$\alpha$\end{document}</tex-math></alternatives></inline-formula> and <inline-formula><alternatives><mml:math id="inf16"><mml:mi>β</mml:mi></mml:math><tex-math id="inft16">\begin{document}$\beta$\end{document}</tex-math></alternatives></inline-formula> that are to be learned from the data such that both the following equations are satisfied:<disp-formula id="equ5"><label>(5)</label><alternatives><mml:math id="m5"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mtable rowspacing="3pt" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true"><mml:mtr><mml:mtd><mml:mtext>mean</mml:mtext><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mtext>TransformedCommonPeaks</mml:mtext><mml:mrow class="MJX-TeXAtom-ORD"><mml:mtext>reference</mml:mtext></mml:mrow></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:mtd><mml:mtd><mml:mi/><mml:mo>=</mml:mo><mml:mtext>mean</mml:mtext><mml:mo stretchy="false">(</mml:mo><mml:mi>α</mml:mi><mml:mo>×</mml:mo><mml:msub><mml:mrow class="MJX-TeXAtom-ORD"><mml:msup><mml:mtext>TransformedCommonPeaks</mml:mtext><mml:mi>β</mml:mi></mml:msup></mml:mrow><mml:mrow class="MJX-TeXAtom-ORD"><mml:mtext>target</mml:mtext></mml:mrow></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:mstyle></mml:math><tex-math id="t5">\begin{document}$$\displaystyle \newcommand{\fref}[1]{Figure~\ref{#1}}\newcommand{\sref}[1]{Section~\ref{#1}} \text{mean}(\text{TransformedCommonPeaks}_{\text{reference}}) &amp;= \text{mean}(\alpha \times {\text{TransformedCommonPeaks}^\beta}_{\text{target}})$$\end{document}</tex-math></alternatives></disp-formula><disp-formula id="equ6"><label>(6)</label><alternatives><mml:math id="m6"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mtable rowspacing="3pt" columnspacing="0em 2em 0em 2em 0em 2em 0em 2em 0em 2em 0em" displaystyle="true"><mml:mtr><mml:mtd><mml:mtext>mean</mml:mtext><mml:mo stretchy="false">(</mml:mo><mml:msub><mml:mtext>TransformedCommonBackground</mml:mtext><mml:mrow class="MJX-TeXAtom-ORD"><mml:mtext>reference</mml:mtext></mml:mrow></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:mtd><mml:mtd><mml:mi/><mml:mo>=</mml:mo><mml:mtext>mean</mml:mtext><mml:mo stretchy="false">(</mml:mo><mml:mi>α</mml:mi><mml:mo>×</mml:mo><mml:msubsup><mml:mtext>TransformedCommonBackground</mml:mtext><mml:mrow class="MJX-TeXAtom-ORD"><mml:mtext>target</mml:mtext></mml:mrow><mml:mi>β</mml:mi></mml:msubsup><mml:mo stretchy="false">)</mml:mo></mml:mtd></mml:mtr></mml:mtable></mml:mstyle></mml:math><tex-math id="t6">\begin{document}$$\displaystyle \newcommand{\fref}[1]{Figure~\ref{#1}}\newcommand{\sref}[1]{Section~\ref{#1}} \text{mean}(\text{TransformedCommonBackground}_{\text{reference}}) &amp;= \text{mean}(\alpha \times \text{TransformedCommonBackground}^\beta_{\text{target}})$$\end{document}</tex-math></alternatives></disp-formula></p><p>Specifically, <italic>α</italic> is a scale factor that shifts the transformed <inline-formula><alternatives><mml:math id="inf17"><mml:mstyle><mml:mrow><mml:mstyle displaystyle="false"><mml:mo>−</mml:mo><mml:msub><mml:mi>log</mml:mi><mml:mrow><mml:mn>10</mml:mn></mml:mrow></mml:msub><mml:mo>⁡</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>p</mml:mi><mml:mtext>-values</mml:mtext></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mstyle></mml:mrow></mml:mstyle></mml:math><tex-math id="inft17">\begin{document}$-\log_{10}({p\text{-values}})$\end{document}</tex-math></alternatives></inline-formula> of the target data set in log scale, and <italic>β</italic> is a power transformation parameter that rotates the transformed <inline-formula><alternatives><mml:math id="inf18"><mml:mstyle><mml:mrow><mml:mstyle displaystyle="false"><mml:mo>−</mml:mo><mml:msub><mml:mi>log</mml:mi><mml:mrow><mml:mn>10</mml:mn></mml:mrow></mml:msub><mml:mo>⁡</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>p</mml:mi><mml:mtext>-values</mml:mtext></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mstyle></mml:mrow></mml:mstyle></mml:math><tex-math id="inft18">\begin{document}$-\log_{10}({p\text{-values}})$\end{document}</tex-math></alternatives></inline-formula> of the target data set in log scale (<xref ref-type="fig" rid="fig2s2">Figure 2—figure supplement 2</xref>). There is one and only one set of values for <italic>α</italic> and <italic>β</italic> that can simultaneously satisfy both the above equations for <italic>common peaks</italic> and the <italic>common background</italic> (<xref ref-type="bibr" rid="bib73">Xiang et al., 2020</xref>).</p><p>The values of <italic>α</italic> and <italic>β</italic> were estimated by the Powell minimization method implemented in scipy (<xref ref-type="bibr" rid="bib14">Fletcher and Powell, 1963</xref>; <xref ref-type="bibr" rid="bib70">Virtanen et al., 2020</xref>). The resulting normalized <inline-formula><alternatives><mml:math id="inf19"><mml:mstyle><mml:mrow><mml:mstyle displaystyle="false"><mml:mo>−</mml:mo><mml:msub><mml:mi>log</mml:mi><mml:mrow><mml:mn>10</mml:mn></mml:mrow></mml:msub><mml:mo>⁡</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>p</mml:mi><mml:mtext>-values</mml:mtext></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mstyle></mml:mrow></mml:mstyle></mml:math><tex-math id="inft19">\begin{document}$-\log_{10}({p\text{-values}})$\end{document}</tex-math></alternatives></inline-formula> were used for all downstream analyses (<xref ref-type="fig" rid="fig2s3">Figure 2—figure supplement 3</xref>).</p></sec><sec id="s4-3"><title>Training endogenous gene expression models</title><p>We trained CNN and ridge regression models, each, to predict gene expression using histone PTM tracks. Input features for each gene were centered at its TSS. We used an input context size of 10,000 base pairs for all analyses subsequent to <xref ref-type="fig" rid="fig3">Figure 3</xref>. For all analyses, we obtained predictions from our models by averaging predictions ensembled across 100 computational replicates.</p><p>To train CNN models, the normalized histone PTM tracks for each gene were processed with successive convolutional blocks. Each convolutional block consisted of a batch-normalization layer, rectified linear units (ReLU), a convolutional layer consisting of 32 convolutional kernels, each of width 5, followed by a dropout with 0.1 probability. Finally, a pooling layer was applied to gradually reduce the dimension of the features. After being processed with five such convolutional blocks, the output was flattened and passed through a fully connected layer consisting of 16 neurons and a ReLU activation. This was ultimately processed with a fully connected layer with a single output and a linear activation (since this was a regression task). The models were trained with a mean squared error loss using the Adam optimizer with a learning rate of 0.001 for the first 50 epochs and 0.0005 for the remaining 50 epochs. Training CNN models took about 1.5 hours on 1 NVIDIA A100 Tensor Core GPU.</p></sec><sec id="s4-4"><title>Interrogating the features learned by CNNs</title><p>To see how different features affected predicted levels of expression, we systematically perturbed each input feature and determined how much the perturbation affected predicted expression levels. To be concrete, we denoted the epigenetic feature at position <inline-formula><alternatives><mml:math id="inf20"><mml:mi>i</mml:mi></mml:math><tex-math id="inft20">\begin{document}$i$\end{document}</tex-math></alternatives></inline-formula> of gene <inline-formula><alternatives><mml:math id="inf21"><mml:mi>g</mml:mi></mml:math><tex-math id="inft21">\begin{document}$g$\end{document}</tex-math></alternatives></inline-formula> in cell type <inline-formula><alternatives><mml:math id="inf22"><mml:mi>C</mml:mi><mml:mi>T</mml:mi></mml:math><tex-math id="inft22">\begin{document}$CT$\end{document}</tex-math></alternatives></inline-formula> as <inline-formula><alternatives><mml:math id="inf23"><mml:msubsup><mml:mi>E</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi>i</mml:mi></mml:mrow><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi>C</mml:mi><mml:mi>T</mml:mi><mml:mo>,</mml:mo><mml:mi>g</mml:mi></mml:mrow></mml:msubsup></mml:math><tex-math id="inft23">\begin{document}$E^{CT,g}_{i}$\end{document}</tex-math></alternatives></inline-formula>. We then defined a perturbation function that added a scalar value of <inline-formula><alternatives><mml:math id="inf24"><mml:msub><mml:mi>λ</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mn>0</mml:mn></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mn>2500</mml:mn></mml:math><tex-math id="inft24">\begin{document}$\lambda_{0}=2500$\end{document}</tex-math></alternatives></inline-formula> to the epigenetic features within 3 bins of a focal position, say, <inline-formula><alternatives><mml:math id="inf25"><mml:mi>j</mml:mi></mml:math><tex-math id="inft25">\begin{document}$j$\end{document}</tex-math></alternatives></inline-formula>:<disp-formula id="equ7"><alternatives><mml:math id="m7"><mml:mrow><mml:mrow><mml:mrow><mml:msub><mml:mi>F</mml:mi><mml:mi>j</mml:mi></mml:msub><mml:mo>⁢</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:msubsup><mml:mi>E</mml:mi><mml:mn>1</mml:mn><mml:mrow><mml:mrow><mml:mi>C</mml:mi><mml:mo>⁢</mml:mo><mml:mi>T</mml:mi></mml:mrow><mml:mo>,</mml:mo><mml:mi>g</mml:mi></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:mi mathvariant="normal">…</mml:mi><mml:mo>,</mml:mo><mml:msubsup><mml:mi>E</mml:mi><mml:mi>W</mml:mi><mml:mrow><mml:mrow><mml:mi>C</mml:mi><mml:mo>⁢</mml:mo><mml:mi>T</mml:mi></mml:mrow><mml:mo>,</mml:mo><mml:mi>g</mml:mi></mml:mrow></mml:msubsup><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mo>:=</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:msubsup><mml:mi>E</mml:mi><mml:mn>1</mml:mn><mml:mrow><mml:mrow><mml:mi>C</mml:mi><mml:mo>⁢</mml:mo><mml:mi>T</mml:mi></mml:mrow><mml:mo>,</mml:mo><mml:mi>g</mml:mi></mml:mrow></mml:msubsup><mml:mo>,</mml:mo><mml:mi mathvariant="normal">…</mml:mi><mml:mo>,</mml:mo><mml:mrow><mml:msubsup><mml:mi>E</mml:mi><mml:mrow><mml:mi>j</mml:mi><mml:mo>-</mml:mo><mml:mn>3</mml:mn></mml:mrow><mml:mrow><mml:mrow><mml:mi>C</mml:mi><mml:mo>⁢</mml:mo><mml:mi>T</mml:mi></mml:mrow><mml:mo>,</mml:mo><mml:mi>g</mml:mi></mml:mrow></mml:msubsup><mml:mo>+</mml:mo><mml:msub><mml:mi>λ</mml:mi><mml:mn>0</mml:mn></mml:msub></mml:mrow><mml:mo>,</mml:mo><mml:mi mathvariant="normal">…</mml:mi><mml:mo>,</mml:mo><mml:mrow><mml:msubsup><mml:mi>E</mml:mi><mml:mrow><mml:mi>j</mml:mi><mml:mo>+</mml:mo><mml:mn>3</mml:mn></mml:mrow><mml:mrow><mml:mrow><mml:mi>C</mml:mi><mml:mo>⁢</mml:mo><mml:mi>T</mml:mi></mml:mrow><mml:mo>,</mml:mo><mml:mi>g</mml:mi></mml:mrow></mml:msubsup><mml:mo>+</mml:mo><mml:msub><mml:mi>λ</mml:mi><mml:mn>0</mml:mn></mml:msub></mml:mrow><mml:mo>,</mml:mo><mml:mi mathvariant="normal">…</mml:mi><mml:mo>,</mml:mo><mml:msubsup><mml:mi>E</mml:mi><mml:mi>W</mml:mi><mml:mrow><mml:mrow><mml:mi>C</mml:mi><mml:mo>⁢</mml:mo><mml:mi>T</mml:mi></mml:mrow><mml:mo>,</mml:mo><mml:mi>g</mml:mi></mml:mrow></mml:msubsup><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mo>,</mml:mo></mml:mrow></mml:math><tex-math id="t7">\begin{document}$$\displaystyle F_{j}\left(E^{CT,g}_{1},\ldots,E^{CT,g}_{W}\right):=\left(E^{CT,g}_{1},\ldots,E^{CT,g}_{j-3}+\lambda_{0},\ldots,E^{CT,g}_{j+3}+\lambda_{0},\ldots,E^{CT,g}_{W}\right),$$\end{document}</tex-math></alternatives></disp-formula></p><p>recalling that <inline-formula><alternatives><mml:math id="inf26"><mml:mi>W</mml:mi></mml:math><tex-math id="inft26">\begin{document}$W$\end{document}</tex-math></alternatives></inline-formula> is the number of bins of 25 base pairs considered by our models, which is set to 401, corresponding to a 10,000 base pair input context length, for all analyses subsequent to <xref ref-type="fig" rid="fig3">Figure 3</xref>. These perturbations corresponded to ∼150 base pairs, which is roughly the length of DNA wrapped around a nucleosome.</p><p>To produce <xref ref-type="fig" rid="fig4">Figure 4</xref>, we applied the above perturbation functions, <inline-formula><alternatives><mml:math id="inf27"><mml:msub><mml:mi>F</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mn>1</mml:mn></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mo>…</mml:mo><mml:mo>,</mml:mo><mml:msub><mml:mi>F</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi>W</mml:mi></mml:mrow></mml:msub></mml:math><tex-math id="inft27">\begin{document}$F_{1},\ldots,F_{W}$\end{document}</tex-math></alternatives></inline-formula> to a histone PTM track of interest, and then measured the fold-change in predicted expression. To account for differences in the endogenous histone PTM tracks between genes, we averaged these fold-changes across 500 randomly chosen genes.</p></sec><sec id="s4-5"><title>In silico modeling of dCas9-p300-based epigenome editing</title><p>Our model of how dCas9-p300 perturbs local histone PTMs has three separate components. We describe each of these components in turn, and then present the full model below. Throughout, we write <inline-formula><alternatives><mml:math id="inf28"><mml:mi>j</mml:mi></mml:math><tex-math id="inft28">\begin{document}$j$\end{document}</tex-math></alternatives></inline-formula> for the position that the gRNA targets.</p><p>First, we modeled steric hindrance of dCas9 due to nucleosomes. We used MNase-seq signal strength as a proxy for nucleosome occupancy. Letting <inline-formula><alternatives><mml:math id="inf29"><mml:msub><mml:mi>m</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:math><tex-math id="inft29">\begin{document}$m_{j}$\end{document}</tex-math></alternatives></inline-formula> be the MNase-seq read coverage at the gRNA binding site, we modeled steric hindrance by scaling the acetylation activity of dCas9-p300 by a factor of <inline-formula><alternatives><mml:math id="inf30"><mml:mi>exp</mml:mi><mml:mo>⁡</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mo>−</mml:mo><mml:mn>5</mml:mn><mml:mo>×</mml:mo><mml:msub><mml:mi>m</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>)</mml:mo></mml:mrow></mml:math><tex-math id="inft30">\begin{document}$\exp\left(-5\times m_{j}\right)$\end{document}</tex-math></alternatives></inline-formula>.</p><p>Second, we assumed that dCas9-p300 primarily alters the levels of H3K27ac only locally. As such, we modeled the acetylation activity of dCas9-p300 at a particular locus as a Gaussian kernel centered at the gRNA. Concretely, the acetylation activity at position <inline-formula><alternatives><mml:math id="inf31"><mml:mi>i</mml:mi></mml:math><tex-math id="inft31">\begin{document}$i$\end{document}</tex-math></alternatives></inline-formula> is multiplied by a factor of <inline-formula><alternatives><mml:math id="inf32"><mml:mi>exp</mml:mi><mml:mo>⁡</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mo>−</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:mi>i</mml:mi><mml:mo>−</mml:mo><mml:mi>j</mml:mi><mml:msup><mml:mo stretchy="false">)</mml:mo><mml:mrow class="MJX-TeXAtom-ORD"><mml:mn>2</mml:mn></mml:mrow></mml:msup><mml:mrow class="MJX-TeXAtom-ORD"><mml:mo>/</mml:mo></mml:mrow><mml:mn>2</mml:mn><mml:msup><mml:mi>σ</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mn>2</mml:mn></mml:mrow></mml:msup><mml:mo>)</mml:mo></mml:mrow></mml:math><tex-math id="inft32">\begin{document}$\exp\left(-(i-j)^{2}/2\sigma^{2}\right)$\end{document}</tex-math></alternatives></inline-formula>, where <inline-formula><alternatives><mml:math id="inf33"><mml:msup><mml:mi>σ</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mn>2</mml:mn></mml:mrow></mml:msup></mml:math><tex-math id="inft33">\begin{document}$\sigma^{2}$\end{document}</tex-math></alternatives></inline-formula> is a parameter of the model.</p><p>Finally, we assumed that dCas9-p300 can only acetylate histones where they currently are – it cannot move histones or increase H3K27ac levels outside of histones. To model this mathematically, we multiplied the acetylation activity at site <inline-formula><alternatives><mml:math id="inf34"><mml:mi>i</mml:mi></mml:math><tex-math id="inft34">\begin{document}$i$\end{document}</tex-math></alternatives></inline-formula> by the MNase read coverage, <inline-formula><alternatives><mml:math id="inf35"><mml:msub><mml:mi>m</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:math><tex-math id="inft35">\begin{document}$m_{i}$\end{document}</tex-math></alternatives></inline-formula>. Therefore, if the MNase read coverage is 0 (i.e., there is no evidence of histones at that locus), then the amount of H3K27ac added to that position is also 0.</p><p>Putting this all together, for a guide targeting at position <inline-formula><alternatives><mml:math id="inf36"><mml:mi>j</mml:mi></mml:math><tex-math id="inft36">\begin{document}$j$\end{document}</tex-math></alternatives></inline-formula>, the effect on H3K27ac levels at position <inline-formula><alternatives><mml:math id="inf37"><mml:mi>i</mml:mi></mml:math><tex-math id="inft37">\begin{document}$i$\end{document}</tex-math></alternatives></inline-formula> is proportional to<disp-formula id="equ8"><alternatives><mml:math id="m8"><mml:mrow><mml:mrow><mml:mi>exp</mml:mi><mml:mo>⁡</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mo>-</mml:mo><mml:mrow><mml:mn>5</mml:mn><mml:mo>⁢</mml:mo><mml:msub><mml:mi>m</mml:mi><mml:mi>j</mml:mi></mml:msub></mml:mrow></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mo>×</mml:mo><mml:mrow><mml:mi>exp</mml:mi><mml:mo>⁡</mml:mo><mml:mrow><mml:mo>[</mml:mo><mml:mfrac><mml:mrow><mml:mo>-</mml:mo><mml:msup><mml:mrow><mml:mo stretchy="false">(</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>-</mml:mo><mml:mi>j</mml:mi></mml:mrow><mml:mo stretchy="false">)</mml:mo></mml:mrow><mml:mn>2</mml:mn></mml:msup></mml:mrow><mml:mrow><mml:mn>2</mml:mn><mml:mo>⁢</mml:mo><mml:msup><mml:mi>σ</mml:mi><mml:mn>2</mml:mn></mml:msup></mml:mrow></mml:mfrac><mml:mo>]</mml:mo></mml:mrow></mml:mrow><mml:mo>×</mml:mo><mml:msub><mml:mi>m</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow></mml:math><tex-math id="t8">\begin{document}$$\displaystyle \exp\left(-5\,m_{j}\right)\times\exp\left[\frac{-(i-j)^{2}}{2\sigma^{2}}\right]\times m_{i}$$\end{document}</tex-math></alternatives></disp-formula></p><p>The constant of proportionality (i.e., how strong we expect dCas9-p300 to be overall) is treated as another free parameter, which we denote by <italic>λ</italic>.</p><p>ENCODE has epigenetic data for the HEK293 cell line, but we performed our dCas9-p300 perturbations in the HEK293T cell line. As such, we used the HEK293 histone PTM as well as RNA-seq data as a stand-in for the HEK293T histone PTM and RNA-seq levels. This substitution is justified as gene expression levels for HEK293 and HEK293T are highly concordant (<xref ref-type="fig" rid="fig3s2">Figure 3—figure supplement 2</xref>). Indeed, the Spearman’s rank correlation between expression levels for HEK293 and two independent measurements of expression levels in HEK293T are 0.86 and 0.88, which are comparable to the correlation between the two independent experiments in HEK293T (<inline-formula><alternatives><mml:math id="inf38"><mml:mi>ρ</mml:mi><mml:mo>=</mml:mo><mml:mn>0.92</mml:mn></mml:math><tex-math id="inft38">\begin{document}$\rho=0.92$\end{document}</tex-math></alternatives></inline-formula>). That is, the correlation across experiments within HEK293T cells is only slightly higher than the correlation between HEK293 and HEK293T, suggesting that cross-cell type differences between HEK293 and HEK293T are on the same order as the inherent experimental and biological noise within a single cell type.</p></sec><sec id="s4-6"><title>Experimental procedure</title><p>The details of dCas9-p300 epigenome editing, qPCR, CUT&amp;RUN, and MNase-seq experiments are provided in the supplementary material.</p></sec></sec></body><back><sec sec-type="additional-information" id="s5"><title>Additional information</title><fn-group content-type="competing-interest"><title>Competing interests</title><fn fn-type="COI-statement" id="conf1"><p>No competing interests declared</p></fn></fn-group><fn-group content-type="author-contribution"><title>Author contributions</title><fn fn-type="con" id="con1"><p>Conceptualization, Data curation, Software, Formal analysis, Validation, Investigation, Visualization, Methodology, Writing – original draft</p></fn><fn fn-type="con" id="con2"><p>Conceptualization, Resources, Data curation, Validation, Investigation, Visualization, Writing – original draft</p></fn><fn fn-type="con" id="con3"><p>Conceptualization, Formal analysis, Investigation, Methodology, Writing – original draft</p></fn><fn fn-type="con" id="con4"><p>Resources, Data curation, Investigation, Writing – review and editing</p></fn><fn fn-type="con" id="con5"><p>Resources, Data curation, Investigation, Writing – review and editing</p></fn><fn fn-type="con" id="con6"><p>Conceptualization, Resources, Supervision, Funding acquisition, Investigation, Writing – original draft, Project administration, Writing – review and editing</p></fn><fn fn-type="con" id="con7"><p>Conceptualization, Supervision, Funding acquisition, Investigation, Methodology, Writing – original draft</p></fn></fn-group></sec><sec sec-type="supplementary-material" id="s6"><title>Additional files</title><supplementary-material id="mdar"><label>MDAR checklist</label><media xlink:href="elife-92991-mdarchecklist1-v1.docx" mimetype="application" mime-subtype="docx"/></supplementary-material></sec><sec sec-type="data-availability" id="s7"><title>Data availability</title><p>The MNase-seq data for the HEK293T cell line is available at BioProject ID PRJNA892960 on SRA. The data from the dCas9-p300 K562 Perturb-seq experiments is available at GSE255610 on SRA. Code and data for training the gene expression models, along with code for generating the figures in the manuscript, are available at <ext-link ext-link-type="uri" xlink:href="https://github.com/songlab-cal/epigenome_editing_2023">https://github.com/songlab-cal/epigenome_editing_2023</ext-link> (copy archived at <xref ref-type="bibr" rid="bib3">Batra, 2023</xref>).</p><p>The following dataset was generated:</p><p><element-citation publication-type="data" specific-use="isSupplementedBy" id="dataset1"><person-group person-group-type="author"><name><surname>Batra</surname><given-names>SS</given-names></name><name><surname>Cabrera</surname><given-names>A</given-names></name><name><surname>Spence</surname><given-names>JP</given-names></name><name><surname>Goell</surname><given-names>J</given-names></name><name><surname>Anand</surname><given-names>SS</given-names></name><name><surname>Hilton</surname><given-names>IB</given-names></name><name><surname>Song</surname><given-names>YS</given-names></name></person-group><year iso-8601-date="2025">2025</year><data-title>MNase-seq data for the HEK293T cell line</data-title><source>European Nucleotide Archive</source><pub-id pub-id-type="accession" xlink:href="https://www.ebi.ac.uk/ena/browser/view/PRJNA892960">PRJNA892960</pub-id></element-citation></p><p>The following previously published dataset was used:</p><p><element-citation publication-type="data" specific-use="references" id="dataset2"><person-group person-group-type="author"><name><surname>Goell</surname><given-names>J</given-names></name><name><surname>Li</surname><given-names>J</given-names></name><name><surname>Mahata</surname><given-names>B</given-names></name><name><surname>Kim</surname><given-names>S</given-names></name><name><surname>Shah</surname><given-names>S</given-names></name><name><surname>Shah</surname><given-names>S</given-names></name><name><surname>Contreras</surname><given-names>M</given-names></name><name><surname>Misra</surname><given-names>S</given-names></name><name><surname>Reed</surname><given-names>D</given-names></name><name><surname>Bedford</surname><given-names>GC</given-names></name><name><surname>Escobar</surname><given-names>M</given-names></name><name><surname>Hilton</surname><given-names>IB</given-names></name></person-group><year iso-8601-date="2025">2025</year><data-title>Tailoring a CRISPR/Cas-based Epigenome Editor for Programmable Chromatin Acylation and Decreased Cytotoxicity</data-title><source>NCBI Gene Expression Omnibus</source><pub-id pub-id-type="accession" xlink:href="https://www.ncbi.nlm.nih.gov/geo/query/acc.cgi?acc=GSE255610">GSE255610</pub-id></element-citation></p></sec><ack id="ack"><title>Acknowledgements</title><p>This research was supported in part by NIH grants R35-GM134922 and R35-GM143532.</p></ack><ref-list><title>References</title><ref id="bib1"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Avsec</surname><given-names>Ž</given-names></name><name><surname>Agarwal</surname><given-names>V</given-names></name><name><surname>Visentin</surname><given-names>D</given-names></name><name><surname>Ledsam</surname><given-names>JR</given-names></name><name><surname>Grabska-Barwinska</surname><given-names>A</given-names></name><name><surname>Taylor</surname><given-names>KR</given-names></name><name><surname>Assael</surname><given-names>Y</given-names></name><name><surname>Jumper</surname><given-names>J</given-names></name><name><surname>Kohli</surname><given-names>P</given-names></name><name><surname>Kelley</surname><given-names>DR</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>Effective gene expression prediction from sequence by integrating long-range interactions</article-title><source>Nature Methods</source><volume>18</volume><fpage>1196</fpage><lpage>1203</lpage><pub-id pub-id-type="doi">10.1038/s41592-021-01252-x</pub-id><pub-id pub-id-type="pmid">34608324</pub-id></element-citation></ref><ref id="bib2"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Barrett</surname><given-names>T</given-names></name><name><surname>Wilhite</surname><given-names>SE</given-names></name><name><surname>Ledoux</surname><given-names>P</given-names></name><name><surname>Evangelista</surname><given-names>C</given-names></name><name><surname>Kim</surname><given-names>IF</given-names></name><name><surname>Tomashevsky</surname><given-names>M</given-names></name><name><surname>Marshall</surname><given-names>KA</given-names></name><name><surname>Phillippy</surname><given-names>KH</given-names></name><name><surname>Sherman</surname><given-names>PM</given-names></name><name><surname>Holko</surname><given-names>M</given-names></name><name><surname>Yefanov</surname><given-names>A</given-names></name><name><surname>Lee</surname><given-names>H</given-names></name><name><surname>Zhang</surname><given-names>N</given-names></name><name><surname>Robertson</surname><given-names>CL</given-names></name><name><surname>Serova</surname><given-names>N</given-names></name><name><surname>Davis</surname><given-names>S</given-names></name><name><surname>Soboleva</surname><given-names>A</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>NCBI GEO: Archive for functional genomics data sets--update</article-title><source>Nucleic Acids Research</source><volume>41</volume><fpage>D991</fpage><lpage>D995</lpage><pub-id pub-id-type="doi">10.1093/nar/gks1193</pub-id><pub-id pub-id-type="pmid">23193258</pub-id></element-citation></ref><ref id="bib3"><element-citation publication-type="software"><person-group person-group-type="author"><name><surname>Batra</surname><given-names>SS</given-names></name></person-group><year iso-8601-date="2023">2023</year><data-title>Epigenome_editing_2023</data-title><version designator="swh:1:rev:ab13aea97b8040ff1fd150a136e268adb4ecc70f">swh:1:rev:ab13aea97b8040ff1fd150a136e268adb4ecc70f</version><source>Software Heritage</source><ext-link ext-link-type="uri" xlink:href="https://archive.softwareheritage.org/swh:1:dir:d182d57dade3921924eb875eab827c85b8fc3731;origin=https://github.com/songlab-cal/epigenome_editing_2023;visit=swh:1:snp:73125247779cf5e6069c3cbd01b81a9748357a96;anchor=swh:1:rev:ab13aea97b8040ff1fd150a136e268adb4ecc70f">https://archive.softwareheritage.org/swh:1:dir:d182d57dade3921924eb875eab827c85b8fc3731;origin=https://github.com/songlab-cal/epigenome_editing_2023;visit=swh:1:snp:73125247779cf5e6069c3cbd01b81a9748357a96;anchor=swh:1:rev:ab13aea97b8040ff1fd150a136e268adb4ecc70f</ext-link></element-citation></ref><ref id="bib4"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Benjamini</surname><given-names>Y</given-names></name><name><surname>Hochberg</surname><given-names>Y</given-names></name></person-group><year iso-8601-date="1995">1995</year><article-title>Controlling the false discovery rate: A practical and powerful approach to multiple testing</article-title><source>Journal of the Royal Statistical Society Series B</source><volume>57</volume><fpage>289</fpage><lpage>300</lpage><pub-id pub-id-type="doi">10.1111/j.2517-6161.1995.tb02031.x</pub-id></element-citation></ref><ref id="bib5"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Chen</surname><given-names>Y</given-names></name><name><surname>Xie</surname><given-names>M</given-names></name><name><surname>Wen</surname><given-names>J</given-names></name></person-group><year iso-8601-date="2022">2022</year><article-title>Predicting gene expression from histone modifications with self-attention based neural networks and transfer learning</article-title><source>Frontiers in Genetics</source><volume>13</volume><elocation-id>1081842</elocation-id><pub-id pub-id-type="doi">10.3389/fgene.2022.1081842</pub-id></element-citation></ref><ref id="bib6"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Cheng</surname><given-names>Y</given-names></name><name><surname>He</surname><given-names>C</given-names></name><name><surname>Wang</surname><given-names>M</given-names></name><name><surname>Ma</surname><given-names>X</given-names></name><name><surname>Mo</surname><given-names>F</given-names></name><name><surname>Yang</surname><given-names>S</given-names></name><name><surname>Han</surname><given-names>J</given-names></name><name><surname>Wei</surname><given-names>X</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Targeting epigenetic regulators for cancer therapy: mechanisms and advances in clinical trials</article-title><source>Signal Transduction and Targeted Therapy</source><volume>4</volume><elocation-id>62</elocation-id><pub-id pub-id-type="doi">10.1038/s41392-019-0095-0</pub-id><pub-id pub-id-type="pmid">31871779</pub-id></element-citation></ref><ref id="bib7"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Concordet</surname><given-names>JP</given-names></name><name><surname>Haeussler</surname><given-names>M</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>CRISPOR: Intuitive guide selection for CRISPR/Cas9 genome editing experiments and screens</article-title><source>Nucleic Acids Research</source><volume>46</volume><fpage>W242</fpage><lpage>W245</lpage><pub-id pub-id-type="doi">10.1093/nar/gky354</pub-id><pub-id pub-id-type="pmid">29762716</pub-id></element-citation></ref><ref id="bib8"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Cui</surname><given-names>K</given-names></name><name><surname>Zhao</surname><given-names>K</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>Genome-wide approaches to determining nucleosome occupancy in metazoans using MNase-Seq</article-title><source>Methods in Molecular Biology</source><volume>833</volume><fpage>413</fpage><lpage>419</lpage><pub-id pub-id-type="doi">10.1007/978-1-61779-477-3_24</pub-id><pub-id pub-id-type="pmid">22183607</pub-id></element-citation></ref><ref id="bib9"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Doench</surname><given-names>JG</given-names></name><name><surname>Fusi</surname><given-names>N</given-names></name><name><surname>Sullender</surname><given-names>M</given-names></name><name><surname>Hegde</surname><given-names>M</given-names></name><name><surname>Vaimberg</surname><given-names>EW</given-names></name><name><surname>Donovan</surname><given-names>KF</given-names></name><name><surname>Smith</surname><given-names>I</given-names></name><name><surname>Tothova</surname><given-names>Z</given-names></name><name><surname>Wilen</surname><given-names>C</given-names></name><name><surname>Orchard</surname><given-names>R</given-names></name><name><surname>Virgin</surname><given-names>HW</given-names></name><name><surname>Listgarten</surname><given-names>J</given-names></name><name><surname>Root</surname><given-names>DE</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Optimized sgRNA design to maximize activity and minimize off-target effects of CRISPR-Cas9</article-title><source>Nature Biotechnology</source><volume>34</volume><fpage>184</fpage><lpage>191</lpage><pub-id pub-id-type="doi">10.1038/nbt.3437</pub-id><pub-id pub-id-type="pmid">26780180</pub-id></element-citation></ref><ref id="bib10"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Dominguez</surname><given-names>AA</given-names></name><name><surname>Chavez</surname><given-names>MG</given-names></name><name><surname>Urke</surname><given-names>A</given-names></name><name><surname>Gao</surname><given-names>Y</given-names></name><name><surname>Wang</surname><given-names>L</given-names></name><name><surname>Qi</surname><given-names>LS</given-names></name></person-group><year iso-8601-date="2022">2022</year><article-title>CRISPR-Mediated synergistic epigenetic and transcriptional control</article-title><source>The CRISPR Journal</source><volume>5</volume><fpage>264</fpage><lpage>275</lpage><pub-id pub-id-type="doi">10.1089/crispr.2021.0099</pub-id><pub-id pub-id-type="pmid">35271371</pub-id></element-citation></ref><ref id="bib11"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Escobar</surname><given-names>M</given-names></name><name><surname>Li</surname><given-names>J</given-names></name><name><surname>Patel</surname><given-names>A</given-names></name><name><surname>Liu</surname><given-names>S</given-names></name><name><surname>Xu</surname><given-names>Q</given-names></name><name><surname>Hilton</surname><given-names>IB</given-names></name></person-group><year iso-8601-date="2022">2022</year><article-title>Quantification of genome editing and transcriptional control capabilities reveals hierarchies among diverse CRISPR/Cas systems in human cells</article-title><source>ACS Synthetic Biology</source><volume>11</volume><fpage>3239</fpage><lpage>3250</lpage><pub-id pub-id-type="doi">10.1021/acssynbio.2c00156</pub-id><pub-id pub-id-type="pmid">36162812</pub-id></element-citation></ref><ref id="bib12"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Estey</surname><given-names>EH</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>Epigenetics in clinical practice: The examples of azacitidine and decitabine in myelodysplasia and acute myeloid leukemia</article-title><source>Leukemia</source><volume>27</volume><fpage>1803</fpage><lpage>1812</lpage><pub-id pub-id-type="doi">10.1038/leu.2013.173</pub-id><pub-id pub-id-type="pmid">23757301</pub-id></element-citation></ref><ref id="bib13"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Feng</surname><given-names>J</given-names></name><name><surname>Liu</surname><given-names>T</given-names></name><name><surname>Qin</surname><given-names>B</given-names></name><name><surname>Zhang</surname><given-names>Y</given-names></name><name><surname>Liu</surname><given-names>XS</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>Identifying ChIP-seq enrichment using MACS</article-title><source>Nature Protocols</source><volume>7</volume><fpage>1728</fpage><lpage>1740</lpage><pub-id pub-id-type="doi">10.1038/nprot.2012.101</pub-id><pub-id pub-id-type="pmid">22936215</pub-id></element-citation></ref><ref id="bib14"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Fletcher</surname><given-names>R</given-names></name><name><surname>Powell</surname><given-names>MJD</given-names></name></person-group><year iso-8601-date="1963">1963</year><article-title>A rapidly convergent descent method for minimization</article-title><source>The Computer Journal</source><volume>6</volume><fpage>163</fpage><lpage>168</lpage><pub-id pub-id-type="doi">10.1093/comjnl/6.2.163</pub-id></element-citation></ref><ref id="bib15"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Frasca</surname><given-names>F</given-names></name><name><surname>Matteucci</surname><given-names>M</given-names></name><name><surname>Leone</surname><given-names>M</given-names></name><name><surname>Morelli</surname><given-names>MJ</given-names></name><name><surname>Masseroli</surname><given-names>M</given-names></name></person-group><year iso-8601-date="2022">2022</year><article-title>Accurate and highly interpretable prediction of gene expression from histone modifications</article-title><source>BMC Bioinformatics</source><volume>23</volume><elocation-id>151</elocation-id><pub-id pub-id-type="doi">10.1186/s12859-022-04687-x</pub-id><pub-id pub-id-type="pmid">35473556</pub-id></element-citation></ref><ref id="bib16"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Gemberling</surname><given-names>MP</given-names></name><name><surname>Siklenka</surname><given-names>K</given-names></name><name><surname>Rodriguez</surname><given-names>E</given-names></name><name><surname>Tonn-Eisinger</surname><given-names>KR</given-names></name><name><surname>Barrera</surname><given-names>A</given-names></name><name><surname>Liu</surname><given-names>F</given-names></name><name><surname>Kantor</surname><given-names>A</given-names></name><name><surname>Li</surname><given-names>L</given-names></name><name><surname>Cigliola</surname><given-names>V</given-names></name><name><surname>Hazlett</surname><given-names>MF</given-names></name><name><surname>Williams</surname><given-names>CA</given-names></name><name><surname>Bartelt</surname><given-names>LC</given-names></name><name><surname>Madigan</surname><given-names>VJ</given-names></name><name><surname>Bodle</surname><given-names>JC</given-names></name><name><surname>Daniels</surname><given-names>H</given-names></name><name><surname>Rouse</surname><given-names>DC</given-names></name><name><surname>Hilton</surname><given-names>IB</given-names></name><name><surname>Asokan</surname><given-names>A</given-names></name><name><surname>Ciofani</surname><given-names>M</given-names></name><name><surname>Poss</surname><given-names>KD</given-names></name><name><surname>Reddy</surname><given-names>TE</given-names></name><name><surname>West</surname><given-names>AE</given-names></name><name><surname>Gersbach</surname><given-names>CA</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>Transgenic mice for in vivo epigenome editing with CRISPR-based systems</article-title><source>Nature Methods</source><volume>18</volume><fpage>965</fpage><lpage>974</lpage><pub-id pub-id-type="doi">10.1038/s41592-021-01207-2</pub-id><pub-id pub-id-type="pmid">34341582</pub-id></element-citation></ref><ref id="bib17"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Gibney</surname><given-names>ER</given-names></name><name><surname>Nolan</surname><given-names>CM</given-names></name></person-group><year iso-8601-date="2010">2010</year><article-title>Epigenetics and gene expression</article-title><source>Heredity</source><volume>105</volume><fpage>4</fpage><lpage>13</lpage><pub-id pub-id-type="doi">10.1038/hdy.2010.54</pub-id><pub-id pub-id-type="pmid">20461105</pub-id></element-citation></ref><ref id="bib18"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Gilbert</surname><given-names>LA</given-names></name><name><surname>Horlbeck</surname><given-names>MA</given-names></name><name><surname>Adamson</surname><given-names>B</given-names></name><name><surname>Villalta</surname><given-names>JE</given-names></name><name><surname>Chen</surname><given-names>Y</given-names></name><name><surname>Whitehead</surname><given-names>EH</given-names></name><name><surname>Guimaraes</surname><given-names>C</given-names></name><name><surname>Panning</surname><given-names>B</given-names></name><name><surname>Ploegh</surname><given-names>HL</given-names></name><name><surname>Bassik</surname><given-names>MC</given-names></name><name><surname>Qi</surname><given-names>LS</given-names></name><name><surname>Kampmann</surname><given-names>M</given-names></name><name><surname>Weissman</surname><given-names>JS</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>Genome-Scale CRISPR-mediated control of gene repression and activation</article-title><source>Cell</source><volume>159</volume><fpage>647</fpage><lpage>661</lpage><pub-id pub-id-type="doi">10.1016/j.cell.2014.09.029</pub-id><pub-id pub-id-type="pmid">25307932</pub-id></element-citation></ref><ref id="bib19"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Goell</surname><given-names>JH</given-names></name><name><surname>Hilton</surname><given-names>IB</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>CRISPR/Cas-based epigenome editing: advances, applications, and clinical utility</article-title><source>Trends in Biotechnology</source><volume>39</volume><fpage>678</fpage><lpage>691</lpage><pub-id pub-id-type="doi">10.1016/j.tibtech.2020.10.012</pub-id><pub-id pub-id-type="pmid">33972106</pub-id></element-citation></ref><ref id="bib20"><element-citation publication-type="preprint"><person-group person-group-type="author"><name><surname>Goell</surname><given-names>JH</given-names></name><name><surname>Li</surname><given-names>J</given-names></name><name><surname>Mahata</surname><given-names>B</given-names></name><name><surname>Ma</surname><given-names>AJ</given-names></name><name><surname>Kim</surname><given-names>S</given-names></name><name><surname>Shah</surname><given-names>S</given-names></name><name><surname>Shah</surname><given-names>S</given-names></name><name><surname>Contreras</surname><given-names>M</given-names></name><name><surname>Misra</surname><given-names>S</given-names></name><name><surname>Reed</surname><given-names>D</given-names></name><name><surname>Bedford</surname><given-names>GC</given-names></name><name><surname>Escobar</surname><given-names>M</given-names></name><name><surname>Hilton</surname><given-names>IB</given-names></name></person-group><year iso-8601-date="2024">2024</year><article-title>Tailoring a CRISPR/Cas-based epigenome editor for programmable chromatin acylation and decreased cytotoxicity</article-title><source>bioRxiv</source><pub-id pub-id-type="doi">10.1101/2024.09.22.611000</pub-id></element-citation></ref><ref id="bib21"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hafner</surname><given-names>A</given-names></name><name><surname>Boettiger</surname><given-names>A</given-names></name></person-group><year iso-8601-date="2023">2023</year><article-title>The spatial organization of transcriptional control</article-title><source>Nature Reviews. Genetics</source><volume>24</volume><fpage>53</fpage><lpage>68</lpage><pub-id pub-id-type="doi">10.1038/s41576-022-00526-0</pub-id><pub-id pub-id-type="pmid">36104547</pub-id></element-citation></ref><ref id="bib22"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hamdy</surname><given-names>R</given-names></name><name><surname>Maghraby</surname><given-names>FA</given-names></name><name><surname>Omar</surname><given-names>YMK</given-names></name></person-group><year iso-8601-date="2022">2022</year><article-title>ConvChrome: Predicting gene expression based on histone modifications using deep learning techniques</article-title><source>Current Bioinformatics</source><volume>17</volume><fpage>273</fpage><lpage>283</lpage><pub-id pub-id-type="doi">10.2174/1574893616666211214110625</pub-id></element-citation></ref><ref id="bib23"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hilton</surname><given-names>IB</given-names></name><name><surname>D’Ippolito</surname><given-names>AM</given-names></name><name><surname>Vockley</surname><given-names>CM</given-names></name><name><surname>Thakore</surname><given-names>PI</given-names></name><name><surname>Crawford</surname><given-names>GE</given-names></name><name><surname>Reddy</surname><given-names>TE</given-names></name><name><surname>Gersbach</surname><given-names>CA</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>Epigenome editing by a CRISPR-Cas9-based acetyltransferase activates genes from promoters and enhancers</article-title><source>Nature Biotechnology</source><volume>33</volume><fpage>510</fpage><lpage>517</lpage><pub-id pub-id-type="doi">10.1038/nbt.3199</pub-id><pub-id pub-id-type="pmid">25849900</pub-id></element-citation></ref><ref id="bib24"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Holoch</surname><given-names>D</given-names></name><name><surname>Moazed</surname><given-names>D</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>RNA-mediated epigenetic regulation of gene expression</article-title><source>Nature Reviews. Genetics</source><volume>16</volume><fpage>71</fpage><lpage>84</lpage><pub-id pub-id-type="doi">10.1038/nrg3863</pub-id><pub-id pub-id-type="pmid">25554358</pub-id></element-citation></ref><ref id="bib25"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Horlbeck</surname><given-names>MA</given-names></name><name><surname>Witkowsky</surname><given-names>LB</given-names></name><name><surname>Guglielmi</surname><given-names>B</given-names></name><name><surname>Replogle</surname><given-names>JM</given-names></name><name><surname>Gilbert</surname><given-names>LA</given-names></name><name><surname>Villalta</surname><given-names>JE</given-names></name><name><surname>Torigoe</surname><given-names>SE</given-names></name><name><surname>Tjian</surname><given-names>R</given-names></name><name><surname>Weissman</surname><given-names>JS</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Nucleosomes impede Cas9 access to DNA in vivo and in vitro</article-title><source>eLife</source><volume>5</volume><elocation-id>e12677</elocation-id><pub-id pub-id-type="doi">10.7554/eLife.12677</pub-id><pub-id pub-id-type="pmid">26987018</pub-id></element-citation></ref><ref id="bib26"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Isaac</surname><given-names>RS</given-names></name><name><surname>Jiang</surname><given-names>F</given-names></name><name><surname>Doudna</surname><given-names>JA</given-names></name><name><surname>Lim</surname><given-names>WA</given-names></name><name><surname>Narlikar</surname><given-names>GJ</given-names></name><name><surname>Almeida</surname><given-names>R</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Nucleosome breathing and remodeling constrain CRISPR-Cas9 function</article-title><source>eLife</source><volume>5</volume><elocation-id>e13450</elocation-id><pub-id pub-id-type="doi">10.7554/eLife.13450</pub-id><pub-id pub-id-type="pmid">27130520</pub-id></element-citation></ref><ref id="bib27"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Jenuwein</surname><given-names>T</given-names></name><name><surname>Allis</surname><given-names>CD</given-names></name></person-group><year iso-8601-date="2001">2001</year><article-title>Translating the histone code</article-title><source>Science</source><volume>293</volume><fpage>1074</fpage><lpage>1080</lpage><pub-id pub-id-type="doi">10.1126/science.1063127</pub-id><pub-id pub-id-type="pmid">11498575</pub-id></element-citation></ref><ref id="bib28"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Jinek</surname><given-names>M</given-names></name><name><surname>Chylinski</surname><given-names>K</given-names></name><name><surname>Fonfara</surname><given-names>I</given-names></name><name><surname>Hauer</surname><given-names>M</given-names></name><name><surname>Doudna</surname><given-names>JA</given-names></name><name><surname>Charpentier</surname><given-names>E</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>A programmable dual-RNA-guided DNA endonuclease in adaptive bacterial immunity</article-title><source>Science</source><volume>337</volume><fpage>816</fpage><lpage>821</lpage><pub-id pub-id-type="doi">10.1126/science.1225829</pub-id><pub-id pub-id-type="pmid">22745249</pub-id></element-citation></ref><ref id="bib29"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Karlić</surname><given-names>R</given-names></name><name><surname>Chung</surname><given-names>HR</given-names></name><name><surname>Lasserre</surname><given-names>J</given-names></name><name><surname>Vlahovicek</surname><given-names>K</given-names></name><name><surname>Vingron</surname><given-names>M</given-names></name></person-group><year iso-8601-date="2010">2010</year><article-title>Histone modification levels are predictive for gene expression</article-title><source>PNAS</source><volume>107</volume><fpage>2926</fpage><lpage>2931</lpage><pub-id pub-id-type="doi">10.1073/pnas.0909344107</pub-id><pub-id pub-id-type="pmid">20133639</pub-id></element-citation></ref><ref id="bib30"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Keung</surname><given-names>AJ</given-names></name><name><surname>Joung</surname><given-names>JK</given-names></name><name><surname>Khalil</surname><given-names>AS</given-names></name><name><surname>Collins</surname><given-names>JJ</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>Chromatin regulation at the frontier of synthetic biology</article-title><source>Nature Reviews. Genetics</source><volume>16</volume><fpage>159</fpage><lpage>171</lpage><pub-id pub-id-type="doi">10.1038/nrg3900</pub-id><pub-id pub-id-type="pmid">25668787</pub-id></element-citation></ref><ref id="bib31"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kimura</surname><given-names>H</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>Histone modifications for human epigenome analysis</article-title><source>Journal of Human Genetics</source><volume>58</volume><fpage>439</fpage><lpage>445</lpage><pub-id pub-id-type="doi">10.1038/jhg.2013.66</pub-id><pub-id pub-id-type="pmid">23739122</pub-id></element-citation></ref><ref id="bib32"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Klann</surname><given-names>TS</given-names></name><name><surname>Black</surname><given-names>JB</given-names></name><name><surname>Chellappan</surname><given-names>M</given-names></name><name><surname>Safi</surname><given-names>A</given-names></name><name><surname>Song</surname><given-names>L</given-names></name><name><surname>Hilton</surname><given-names>IB</given-names></name><name><surname>Crawford</surname><given-names>GE</given-names></name><name><surname>Reddy</surname><given-names>TE</given-names></name><name><surname>Gersbach</surname><given-names>CA</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>CRISPR-Cas9 epigenome editing enables high-throughput screening for functional regulatory elements in the human genome</article-title><source>Nature Biotechnology</source><volume>35</volume><fpage>561</fpage><lpage>568</lpage><pub-id pub-id-type="doi">10.1038/nbt.3853</pub-id><pub-id pub-id-type="pmid">28369033</pub-id></element-citation></ref><ref id="bib33"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Klemm</surname><given-names>SL</given-names></name><name><surname>Shipony</surname><given-names>Z</given-names></name><name><surname>Greenleaf</surname><given-names>WJ</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Chromatin accessibility and the regulatory epigenome</article-title><source>Nature Reviews. Genetics</source><volume>20</volume><fpage>207</fpage><lpage>220</lpage><pub-id pub-id-type="doi">10.1038/s41576-018-0089-8</pub-id><pub-id pub-id-type="pmid">30675018</pub-id></element-citation></ref><ref id="bib34"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kouzarides</surname><given-names>T</given-names></name></person-group><year iso-8601-date="2007">2007</year><article-title>Chromatin modifications and their function</article-title><source>Cell</source><volume>128</volume><fpage>693</fpage><lpage>705</lpage><pub-id pub-id-type="doi">10.1016/j.cell.2007.02.005</pub-id><pub-id pub-id-type="pmid">17320507</pub-id></element-citation></ref><ref id="bib35"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kundaje</surname><given-names>A</given-names></name><name><surname>Meuleman</surname><given-names>W</given-names></name><name><surname>Ernst</surname><given-names>J</given-names></name><name><surname>Bilenky</surname><given-names>M</given-names></name><name><surname>Yen</surname><given-names>A</given-names></name><name><surname>Heravi-Moussavi</surname><given-names>A</given-names></name><name><surname>Kheradpour</surname><given-names>P</given-names></name><name><surname>Zhang</surname><given-names>Z</given-names></name><name><surname>Wang</surname><given-names>J</given-names></name><name><surname>Ziller</surname><given-names>MJ</given-names></name><name><surname>Amin</surname><given-names>V</given-names></name><name><surname>Whitaker</surname><given-names>JW</given-names></name><name><surname>Schultz</surname><given-names>MD</given-names></name><name><surname>Ward</surname><given-names>LD</given-names></name><name><surname>Sarkar</surname><given-names>A</given-names></name><name><surname>Quon</surname><given-names>G</given-names></name><name><surname>Sandstrom</surname><given-names>RS</given-names></name><name><surname>Eaton</surname><given-names>ML</given-names></name><name><surname>Wu</surname><given-names>Y-C</given-names></name><name><surname>Pfenning</surname><given-names>AR</given-names></name><name><surname>Wang</surname><given-names>X</given-names></name><name><surname>Claussnitzer</surname><given-names>M</given-names></name><name><surname>Liu</surname><given-names>Y</given-names></name><name><surname>Coarfa</surname><given-names>C</given-names></name><name><surname>Harris</surname><given-names>RA</given-names></name><name><surname>Shoresh</surname><given-names>N</given-names></name><name><surname>Epstein</surname><given-names>CB</given-names></name><name><surname>Gjoneska</surname><given-names>E</given-names></name><name><surname>Leung</surname><given-names>D</given-names></name><name><surname>Xie</surname><given-names>W</given-names></name><name><surname>Hawkins</surname><given-names>RD</given-names></name><name><surname>Lister</surname><given-names>R</given-names></name><name><surname>Hong</surname><given-names>C</given-names></name><name><surname>Gascard</surname><given-names>P</given-names></name><name><surname>Mungall</surname><given-names>AJ</given-names></name><name><surname>Moore</surname><given-names>R</given-names></name><name><surname>Chuah</surname><given-names>E</given-names></name><name><surname>Tam</surname><given-names>A</given-names></name><name><surname>Canfield</surname><given-names>TK</given-names></name><name><surname>Hansen</surname><given-names>RS</given-names></name><name><surname>Kaul</surname><given-names>R</given-names></name><name><surname>Sabo</surname><given-names>PJ</given-names></name><name><surname>Bansal</surname><given-names>MS</given-names></name><name><surname>Carles</surname><given-names>A</given-names></name><name><surname>Dixon</surname><given-names>JR</given-names></name><name><surname>Farh</surname><given-names>K-H</given-names></name><name><surname>Feizi</surname><given-names>S</given-names></name><name><surname>Karlic</surname><given-names>R</given-names></name><name><surname>Kim</surname><given-names>A-R</given-names></name><name><surname>Kulkarni</surname><given-names>A</given-names></name><name><surname>Li</surname><given-names>D</given-names></name><name><surname>Lowdon</surname><given-names>R</given-names></name><name><surname>Elliott</surname><given-names>G</given-names></name><name><surname>Mercer</surname><given-names>TR</given-names></name><name><surname>Neph</surname><given-names>SJ</given-names></name><name><surname>Onuchic</surname><given-names>V</given-names></name><name><surname>Polak</surname><given-names>P</given-names></name><name><surname>Rajagopal</surname><given-names>N</given-names></name><name><surname>Ray</surname><given-names>P</given-names></name><name><surname>Sallari</surname><given-names>RC</given-names></name><name><surname>Siebenthall</surname><given-names>KT</given-names></name><name><surname>Sinnott-Armstrong</surname><given-names>NA</given-names></name><name><surname>Stevens</surname><given-names>M</given-names></name><name><surname>Thurman</surname><given-names>RE</given-names></name><name><surname>Wu</surname><given-names>J</given-names></name><name><surname>Zhang</surname><given-names>B</given-names></name><name><surname>Zhou</surname><given-names>X</given-names></name><name><surname>Beaudet</surname><given-names>AE</given-names></name><name><surname>Boyer</surname><given-names>LA</given-names></name><name><surname>De Jager</surname><given-names>PL</given-names></name><name><surname>Farnham</surname><given-names>PJ</given-names></name><name><surname>Fisher</surname><given-names>SJ</given-names></name><name><surname>Haussler</surname><given-names>D</given-names></name><name><surname>Jones</surname><given-names>SJM</given-names></name><name><surname>Li</surname><given-names>W</given-names></name><name><surname>Marra</surname><given-names>MA</given-names></name><name><surname>McManus</surname><given-names>MT</given-names></name><name><surname>Sunyaev</surname><given-names>S</given-names></name><name><surname>Thomson</surname><given-names>JA</given-names></name><name><surname>Tlsty</surname><given-names>TD</given-names></name><name><surname>Tsai</surname><given-names>L-H</given-names></name><name><surname>Wang</surname><given-names>W</given-names></name><name><surname>Waterland</surname><given-names>RA</given-names></name><name><surname>Zhang</surname><given-names>MQ</given-names></name><name><surname>Chadwick</surname><given-names>LH</given-names></name><name><surname>Bernstein</surname><given-names>BE</given-names></name><name><surname>Costello</surname><given-names>JF</given-names></name><name><surname>Ecker</surname><given-names>JR</given-names></name><name><surname>Hirst</surname><given-names>M</given-names></name><name><surname>Meissner</surname><given-names>A</given-names></name><name><surname>Milosavljevic</surname><given-names>A</given-names></name><name><surname>Ren</surname><given-names>B</given-names></name><name><surname>Stamatoyannopoulos</surname><given-names>JA</given-names></name><name><surname>Wang</surname><given-names>T</given-names></name><name><surname>Kellis</surname><given-names>M</given-names></name><collab>Roadmap Epigenomics Consortium</collab></person-group><year iso-8601-date="2015">2015</year><article-title>Integrative analysis of 111 reference human epigenomes</article-title><source>Nature</source><volume>518</volume><fpage>317</fpage><lpage>330</lpage><pub-id pub-id-type="doi">10.1038/nature14248</pub-id><pub-id pub-id-type="pmid">25693563</pub-id></element-citation></ref><ref id="bib36"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kwon</surname><given-names>DY</given-names></name><name><surname>Zhao</surname><given-names>YT</given-names></name><name><surname>Lamonica</surname><given-names>JM</given-names></name><name><surname>Zhou</surname><given-names>Z</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>Locus-specific histone deacetylation using a synthetic CRISPR-Cas9-based HDAC</article-title><source>Nature Communications</source><volume>8</volume><elocation-id>15315</elocation-id><pub-id pub-id-type="doi">10.1038/ncomms15315</pub-id><pub-id pub-id-type="pmid">28497787</pub-id></element-citation></ref><ref id="bib37"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Li</surname><given-names>J</given-names></name><name><surname>Mahata</surname><given-names>B</given-names></name><name><surname>Escobar</surname><given-names>M</given-names></name><name><surname>Goell</surname><given-names>J</given-names></name><name><surname>Wang</surname><given-names>K</given-names></name><name><surname>Khemka</surname><given-names>P</given-names></name><name><surname>Hilton</surname><given-names>IB</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>Programmable human histone phosphorylation and gene activation using a CRISPR/Cas9-based chromatin kinase</article-title><source>Nature Communications</source><volume>12</volume><fpage>1</fpage><lpage>10</lpage><pub-id pub-id-type="doi">10.1038/s41467-021-21188-2</pub-id></element-citation></ref><ref id="bib38"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname><given-names>G</given-names></name><name><surname>Zhang</surname><given-names>Y</given-names></name><name><surname>Zhang</surname><given-names>T</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>Computational approaches for effective CRISPR guide RNA design and evaluation</article-title><source>Computational and Structural Biotechnology Journal</source><volume>18</volume><fpage>35</fpage><lpage>44</lpage><pub-id pub-id-type="doi">10.1016/j.csbj.2019.11.006</pub-id><pub-id pub-id-type="pmid">31890142</pub-id></element-citation></ref><ref id="bib39"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Mahata</surname><given-names>B</given-names></name><name><surname>Cabrera</surname><given-names>A</given-names></name><name><surname>Brenner</surname><given-names>DA</given-names></name><name><surname>Guerra-Resendez</surname><given-names>RS</given-names></name><name><surname>Li</surname><given-names>J</given-names></name><name><surname>Goell</surname><given-names>J</given-names></name><name><surname>Wang</surname><given-names>K</given-names></name><name><surname>Guo</surname><given-names>Y</given-names></name><name><surname>Escobar</surname><given-names>M</given-names></name><name><surname>Parthasarathy</surname><given-names>AK</given-names></name><name><surname>Szadowski</surname><given-names>H</given-names></name><name><surname>Bedford</surname><given-names>G</given-names></name><name><surname>Reed</surname><given-names>DR</given-names></name><name><surname>Kim</surname><given-names>S</given-names></name><name><surname>Hilton</surname><given-names>IB</given-names></name></person-group><year iso-8601-date="2023">2023</year><article-title>Compact engineered human mechanosensitive transactivation modules enable potent and versatile synthetic transcriptional control</article-title><source>Nature Methods</source><volume>20</volume><fpage>1716</fpage><lpage>1728</lpage><pub-id pub-id-type="doi">10.1038/s41592-023-02036-1</pub-id><pub-id pub-id-type="pmid">37813990</pub-id></element-citation></ref><ref id="bib40"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Makasheva</surname><given-names>K</given-names></name><name><surname>Bryan</surname><given-names>LC</given-names></name><name><surname>Anders</surname><given-names>C</given-names></name><name><surname>Panikulam</surname><given-names>S</given-names></name><name><surname>Jinek</surname><given-names>M</given-names></name><name><surname>Fierz</surname><given-names>B</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>Multiplexed single-molecule experiments reveal nucleosome invasion dynamics of the Cas9 genome editor</article-title><source>Journal of the American Chemical Society</source><volume>143</volume><fpage>16313</fpage><lpage>16319</lpage><pub-id pub-id-type="doi">10.1021/jacs.1c06195</pub-id><pub-id pub-id-type="pmid">34597515</pub-id></element-citation></ref><ref id="bib41"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Mali</surname><given-names>P</given-names></name><name><surname>Esvelt</surname><given-names>KM</given-names></name><name><surname>Church</surname><given-names>GM</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>Cas9 as a versatile tool for engineering biology</article-title><source>Nature Methods</source><volume>10</volume><fpage>957</fpage><lpage>963</lpage><pub-id pub-id-type="doi">10.1038/nmeth.2649</pub-id><pub-id pub-id-type="pmid">24076990</pub-id></element-citation></ref><ref id="bib42"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Matharu</surname><given-names>N</given-names></name><name><surname>Ahituv</surname><given-names>N</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>Modulating gene regulation to treat genetic disorders</article-title><source>Nature Reviews. Drug Discovery</source><volume>19</volume><fpage>757</fpage><lpage>775</lpage><pub-id pub-id-type="doi">10.1038/s41573-020-0083-7</pub-id><pub-id pub-id-type="pmid">33020616</pub-id></element-citation></ref><ref id="bib43"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>McKnight</surname><given-names>LE</given-names></name><name><surname>Crandall</surname><given-names>JG</given-names></name><name><surname>Bailey</surname><given-names>TB</given-names></name><name><surname>Banks</surname><given-names>OGB</given-names></name><name><surname>Orlandi</surname><given-names>KN</given-names></name><name><surname>Truong</surname><given-names>VN</given-names></name><name><surname>Donovan</surname><given-names>DA</given-names></name><name><surname>Waddell</surname><given-names>GL</given-names></name><name><surname>Wiles</surname><given-names>ET</given-names></name><name><surname>Hansen</surname><given-names>SD</given-names></name><name><surname>Selker</surname><given-names>EU</given-names></name><name><surname>McKnight</surname><given-names>JN</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>Rapid and inexpensive preparation of genome-wide nucleosome footprints from model and non-model organisms</article-title><source>STAR Protocols</source><volume>2</volume><elocation-id>100486</elocation-id><pub-id pub-id-type="doi">10.1016/j.xpro.2021.100486</pub-id><pub-id pub-id-type="pmid">34041500</pub-id></element-citation></ref><ref id="bib44"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Millán-Zambrano</surname><given-names>G</given-names></name><name><surname>Burton</surname><given-names>A</given-names></name><name><surname>Bannister</surname><given-names>AJ</given-names></name><name><surname>Schneider</surname><given-names>R</given-names></name></person-group><year iso-8601-date="2022">2022</year><article-title>Histone post-translational modifications - cause and consequence of genome function</article-title><source>Nature Reviews. Genetics</source><volume>23</volume><fpage>563</fpage><lpage>580</lpage><pub-id pub-id-type="doi">10.1038/s41576-022-00468-7</pub-id><pub-id pub-id-type="pmid">35338361</pub-id></element-citation></ref><ref id="bib45"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Mohr</surname><given-names>SE</given-names></name><name><surname>Hu</surname><given-names>Y</given-names></name><name><surname>Ewen-Campen</surname><given-names>B</given-names></name><name><surname>Housden</surname><given-names>BE</given-names></name><name><surname>Viswanatha</surname><given-names>R</given-names></name><name><surname>Perrimon</surname><given-names>N</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>CRISPR guide RNA design for research applications</article-title><source>The FEBS Journal</source><volume>283</volume><fpage>3232</fpage><lpage>3238</lpage><pub-id pub-id-type="doi">10.1111/febs.13777</pub-id><pub-id pub-id-type="pmid">27276584</pub-id></element-citation></ref><ref id="bib46"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Nuñez</surname><given-names>JK</given-names></name><name><surname>Chen</surname><given-names>J</given-names></name><name><surname>Pommier</surname><given-names>GC</given-names></name><name><surname>Cogan</surname><given-names>JZ</given-names></name><name><surname>Replogle</surname><given-names>JM</given-names></name><name><surname>Adriaens</surname><given-names>C</given-names></name><name><surname>Ramadoss</surname><given-names>GN</given-names></name><name><surname>Shi</surname><given-names>Q</given-names></name><name><surname>Hung</surname><given-names>KL</given-names></name><name><surname>Samelson</surname><given-names>AJ</given-names></name><name><surname>Pogson</surname><given-names>AN</given-names></name><name><surname>Kim</surname><given-names>JYS</given-names></name><name><surname>Chung</surname><given-names>A</given-names></name><name><surname>Leonetti</surname><given-names>MD</given-names></name><name><surname>Chang</surname><given-names>HY</given-names></name><name><surname>Kampmann</surname><given-names>M</given-names></name><name><surname>Bernstein</surname><given-names>BE</given-names></name><name><surname>Hovestadt</surname><given-names>V</given-names></name><name><surname>Gilbert</surname><given-names>LA</given-names></name><name><surname>Weissman</surname><given-names>JS</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>Genome-wide programmable transcriptional memory by CRISPR-based epigenome editing</article-title><source>Cell</source><volume>184</volume><fpage>2503</fpage><lpage>2519</lpage><pub-id pub-id-type="doi">10.1016/j.cell.2021.03.025</pub-id><pub-id pub-id-type="pmid">33838111</pub-id></element-citation></ref><ref id="bib47"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>O’Geen</surname><given-names>H</given-names></name><name><surname>Ren</surname><given-names>C</given-names></name><name><surname>Nicolet</surname><given-names>CM</given-names></name><name><surname>Perez</surname><given-names>AA</given-names></name><name><surname>Halmai</surname><given-names>J</given-names></name><name><surname>Le</surname><given-names>VM</given-names></name><name><surname>Mackay</surname><given-names>JP</given-names></name><name><surname>Farnham</surname><given-names>PJ</given-names></name><name><surname>Segal</surname><given-names>DJ</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>dCas9-based epigenome editing suggests acquisition of histone methylation is not sufficient for target gene repression</article-title><source>Nucleic Acids Research</source><volume>45</volume><fpage>9901</fpage><lpage>9916</lpage><pub-id pub-id-type="doi">10.1093/nar/gkx578</pub-id><pub-id pub-id-type="pmid">28973434</pub-id></element-citation></ref><ref id="bib48"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Radzisheuskaya</surname><given-names>A</given-names></name><name><surname>Shlyueva</surname><given-names>D</given-names></name><name><surname>Müller</surname><given-names>I</given-names></name><name><surname>Helin</surname><given-names>K</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Optimizing sgRNA position markedly improves the efficiency of CRISPR/dCas9-mediated transcriptional repression</article-title><source>Nucleic Acids Research</source><volume>44</volume><elocation-id>e141</elocation-id><pub-id pub-id-type="doi">10.1093/nar/gkw583</pub-id><pub-id pub-id-type="pmid">27353328</pub-id></element-citation></ref><ref id="bib49"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Rao</surname><given-names>SSP</given-names></name><name><surname>Huntley</surname><given-names>MH</given-names></name><name><surname>Durand</surname><given-names>NC</given-names></name><name><surname>Stamenova</surname><given-names>EK</given-names></name><name><surname>Bochkov</surname><given-names>ID</given-names></name><name><surname>Robinson</surname><given-names>JT</given-names></name><name><surname>Sanborn</surname><given-names>AL</given-names></name><name><surname>Machol</surname><given-names>I</given-names></name><name><surname>Omer</surname><given-names>AD</given-names></name><name><surname>Lander</surname><given-names>ES</given-names></name><name><surname>Aiden</surname><given-names>EL</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>A 3D map of the human genome at kilobase resolution reveals principles of chromatin looping</article-title><source>Cell</source><volume>159</volume><fpage>1665</fpage><lpage>1680</lpage><pub-id pub-id-type="doi">10.1016/j.cell.2014.11.021</pub-id><pub-id pub-id-type="pmid">25497547</pub-id></element-citation></ref><ref id="bib50"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Reik</surname><given-names>W</given-names></name><name><surname>Dean</surname><given-names>W</given-names></name><name><surname>Walter</surname><given-names>J</given-names></name></person-group><year iso-8601-date="2001">2001</year><article-title>Epigenetic reprogramming in mammalian development</article-title><source>Science</source><volume>293</volume><fpage>1089</fpage><lpage>1093</lpage><pub-id pub-id-type="doi">10.1126/science.1063443</pub-id><pub-id pub-id-type="pmid">11498579</pub-id></element-citation></ref><ref id="bib51"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Sanson</surname><given-names>KR</given-names></name><name><surname>Hanna</surname><given-names>RE</given-names></name><name><surname>Hegde</surname><given-names>M</given-names></name><name><surname>Donovan</surname><given-names>KF</given-names></name><name><surname>Strand</surname><given-names>C</given-names></name><name><surname>Sullender</surname><given-names>ME</given-names></name><name><surname>Vaimberg</surname><given-names>EW</given-names></name><name><surname>Goodale</surname><given-names>A</given-names></name><name><surname>Root</surname><given-names>DE</given-names></name><name><surname>Piccioni</surname><given-names>F</given-names></name><name><surname>Doench</surname><given-names>JG</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Optimized libraries for CRISPR-Cas9 genetic screens with multiple modalities</article-title><source>Nature Communications</source><volume>9</volume><elocation-id>5416</elocation-id><pub-id pub-id-type="doi">10.1038/s41467-018-07901-8</pub-id><pub-id pub-id-type="pmid">30575746</pub-id></element-citation></ref><ref id="bib52"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Schmidt</surname><given-names>F</given-names></name><name><surname>Gasparoni</surname><given-names>N</given-names></name><name><surname>Gasparoni</surname><given-names>G</given-names></name><name><surname>Gianmoena</surname><given-names>K</given-names></name><name><surname>Cadenas</surname><given-names>C</given-names></name><name><surname>Polansky</surname><given-names>JK</given-names></name><name><surname>Ebert</surname><given-names>P</given-names></name><name><surname>Nordström</surname><given-names>K</given-names></name><name><surname>Barann</surname><given-names>M</given-names></name><name><surname>Sinha</surname><given-names>A</given-names></name><name><surname>Fröhler</surname><given-names>S</given-names></name><name><surname>Xiong</surname><given-names>J</given-names></name><name><surname>Dehghani Amirabad</surname><given-names>A</given-names></name><name><surname>Behjati Ardakani</surname><given-names>F</given-names></name><name><surname>Hutter</surname><given-names>B</given-names></name><name><surname>Zipprich</surname><given-names>G</given-names></name><name><surname>Felder</surname><given-names>B</given-names></name><name><surname>Eils</surname><given-names>J</given-names></name><name><surname>Brors</surname><given-names>B</given-names></name><name><surname>Chen</surname><given-names>W</given-names></name><name><surname>Hengstler</surname><given-names>JG</given-names></name><name><surname>Hamann</surname><given-names>A</given-names></name><name><surname>Lengauer</surname><given-names>T</given-names></name><name><surname>Rosenstiel</surname><given-names>P</given-names></name><name><surname>Walter</surname><given-names>J</given-names></name><name><surname>Schulz</surname><given-names>MH</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>Combining transcription factor binding affinities with open-chromatin data for accurate gene expression prediction</article-title><source>Nucleic Acids Research</source><volume>45</volume><fpage>54</fpage><lpage>66</lpage><pub-id pub-id-type="doi">10.1093/nar/gkw1061</pub-id><pub-id pub-id-type="pmid">27899623</pub-id></element-citation></ref><ref id="bib53"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Schmidt</surname><given-names>F</given-names></name><name><surname>Kern</surname><given-names>F</given-names></name><name><surname>Schulz</surname><given-names>MH</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>Integrative prediction of gene expression with chromatin accessibility and conformation data</article-title><source>Epigenetics &amp; Chromatin</source><volume>13</volume><elocation-id>4</elocation-id><pub-id pub-id-type="doi">10.1186/s13072-020-0327-0</pub-id><pub-id pub-id-type="pmid">32029002</pub-id></element-citation></ref><ref id="bib54"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Schmidt</surname><given-names>R</given-names></name><name><surname>Steinhart</surname><given-names>Z</given-names></name><name><surname>Layeghi</surname><given-names>M</given-names></name><name><surname>Freimer</surname><given-names>JW</given-names></name><name><surname>Bueno</surname><given-names>R</given-names></name><name><surname>Nguyen</surname><given-names>VQ</given-names></name><name><surname>Blaeschke</surname><given-names>F</given-names></name><name><surname>Ye</surname><given-names>CJ</given-names></name><name><surname>Marson</surname><given-names>A</given-names></name></person-group><year iso-8601-date="2022">2022</year><article-title>CRISPR activation and interference screens decode stimulation responses in primary human T cells</article-title><source>Science</source><volume>375</volume><elocation-id>eabj4008</elocation-id><pub-id pub-id-type="doi">10.1126/science.abj4008</pub-id><pub-id pub-id-type="pmid">35113687</pub-id></element-citation></ref><ref id="bib55"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Schreiber</surname><given-names>J</given-names></name><name><surname>Bilmes</surname><given-names>J</given-names></name><name><surname>Noble</surname><given-names>WS</given-names></name></person-group><year iso-8601-date="2020">2020a</year><article-title>Completing the ENCODE3 compendium yields accurate imputations across a variety of assays and human biosamples</article-title><source>Genome Biology</source><volume>21</volume><elocation-id>82</elocation-id><pub-id pub-id-type="doi">10.1186/s13059-020-01978-5</pub-id><pub-id pub-id-type="pmid">32228713</pub-id></element-citation></ref><ref id="bib56"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Schreiber</surname><given-names>J</given-names></name><name><surname>Durham</surname><given-names>T</given-names></name><name><surname>Bilmes</surname><given-names>J</given-names></name><name><surname>Noble</surname><given-names>WS</given-names></name></person-group><year iso-8601-date="2020">2020b</year><article-title>Avocado: a multi-scale deep tensor factorization method learns a latent representation of the human epigenome</article-title><source>Genome Biology</source><volume>21</volume><elocation-id>81</elocation-id><pub-id pub-id-type="doi">10.1186/s13059-020-01977-6</pub-id><pub-id pub-id-type="pmid">32228704</pub-id></element-citation></ref><ref id="bib57"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Segelle</surname><given-names>A</given-names></name><name><surname>Núñez-Álvarez</surname><given-names>Y</given-names></name><name><surname>Oldfield</surname><given-names>AJ</given-names></name><name><surname>Webb</surname><given-names>KM</given-names></name><name><surname>Voigt</surname><given-names>P</given-names></name><name><surname>Luco</surname><given-names>RF</given-names></name></person-group><year iso-8601-date="2022">2022</year><article-title>Histone marks regulate the epithelial-to-mesenchymal transition via alternative splicing</article-title><source>Cell Reports</source><volume>38</volume><elocation-id>110357</elocation-id><pub-id pub-id-type="doi">10.1016/j.celrep.2022.110357</pub-id><pub-id pub-id-type="pmid">35172149</pub-id></element-citation></ref><ref id="bib58"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Sekhon</surname><given-names>A</given-names></name><name><surname>Singh</surname><given-names>R</given-names></name><name><surname>Qi</surname><given-names>Y</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>DeepDiff: DEEP-learning for predicting DIFFerential gene expression from histone modifications</article-title><source>Bioinformatics</source><volume>34</volume><fpage>i891</fpage><lpage>i900</lpage><pub-id pub-id-type="doi">10.1093/bioinformatics/bty612</pub-id><pub-id pub-id-type="pmid">30423076</pub-id></element-citation></ref><ref id="bib59"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Shalem</surname><given-names>O</given-names></name><name><surname>Sanjana</surname><given-names>NE</given-names></name><name><surname>Hartenian</surname><given-names>E</given-names></name><name><surname>Shi</surname><given-names>X</given-names></name><name><surname>Scott</surname><given-names>DA</given-names></name><name><surname>Mikkelson</surname><given-names>T</given-names></name><name><surname>Heckl</surname><given-names>D</given-names></name><name><surname>Ebert</surname><given-names>BL</given-names></name><name><surname>Root</surname><given-names>DE</given-names></name><name><surname>Doench</surname><given-names>JG</given-names></name><name><surname>Zhang</surname><given-names>F</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>Genome-scale CRISPR-Cas9 knockout screening in human cells</article-title><source>Science</source><volume>343</volume><fpage>84</fpage><lpage>87</lpage><pub-id pub-id-type="doi">10.1126/science.1247005</pub-id><pub-id pub-id-type="pmid">24336571</pub-id></element-citation></ref><ref id="bib60"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Singh</surname><given-names>R</given-names></name><name><surname>Lanchantin</surname><given-names>J</given-names></name><name><surname>Robins</surname><given-names>G</given-names></name><name><surname>Qi</surname><given-names>Y</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>DeepChrome: deep-learning for predicting gene expression from histone modifications</article-title><source>Bioinformatics</source><volume>32</volume><fpage>i639</fpage><lpage>i648</lpage><pub-id pub-id-type="doi">10.1093/bioinformatics/btw427</pub-id><pub-id pub-id-type="pmid">27587684</pub-id></element-citation></ref><ref id="bib61"><element-citation publication-type="confproc"><person-group person-group-type="author"><name><surname>Singh</surname><given-names>R</given-names></name><name><surname>Lanchantin</surname><given-names>J</given-names></name><name><surname>Sekhon</surname><given-names>A</given-names></name><name><surname>Qi</surname><given-names>Y</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>Attend and predict: Understanding gene regulation by selective attention on chromatin</article-title><conf-name>Advances in Neural Information Processing Systems</conf-name><fpage>6785</fpage><lpage>6795</lpage><pub-id pub-id-type="pmid">30147283</pub-id></element-citation></ref><ref id="bib62"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Stepper</surname><given-names>P</given-names></name><name><surname>Kungulovski</surname><given-names>G</given-names></name><name><surname>Jurkowska</surname><given-names>RZ</given-names></name><name><surname>Chandra</surname><given-names>T</given-names></name><name><surname>Krueger</surname><given-names>F</given-names></name><name><surname>Reinhardt</surname><given-names>R</given-names></name><name><surname>Reik</surname><given-names>W</given-names></name><name><surname>Jeltsch</surname><given-names>A</given-names></name><name><surname>Jurkowski</surname><given-names>TP</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>Efficient targeted DNA methylation with chimeric dCas9-Dnmt3a-Dnmt3L methyltransferase</article-title><source>Nucleic Acids Research</source><volume>45</volume><fpage>1703</fpage><lpage>1713</lpage><pub-id pub-id-type="doi">10.1093/nar/gkw1112</pub-id><pub-id pub-id-type="pmid">27899645</pub-id></element-citation></ref><ref id="bib63"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Stillman</surname><given-names>B</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Histone modifications: Insights into their influence on gene expression</article-title><source>Cell</source><volume>175</volume><fpage>6</fpage><lpage>9</lpage><pub-id pub-id-type="doi">10.1016/j.cell.2018.08.032</pub-id><pub-id pub-id-type="pmid">30217360</pub-id></element-citation></ref><ref id="bib64"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Strahl</surname><given-names>BD</given-names></name><name><surname>Allis</surname><given-names>CD</given-names></name></person-group><year iso-8601-date="2000">2000</year><article-title>The language of covalent histone modifications</article-title><source>Nature</source><volume>403</volume><fpage>41</fpage><lpage>45</lpage><pub-id pub-id-type="doi">10.1038/47412</pub-id></element-citation></ref><ref id="bib65"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Stricker</surname><given-names>SH</given-names></name><name><surname>Köferle</surname><given-names>A</given-names></name><name><surname>Beck</surname><given-names>S</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>From profiles to function in epigenomics</article-title><source>Nature Reviews. Genetics</source><volume>18</volume><fpage>51</fpage><lpage>66</lpage><pub-id pub-id-type="doi">10.1038/nrg.2016.138</pub-id><pub-id pub-id-type="pmid">27867193</pub-id></element-citation></ref><ref id="bib66"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Swaminathan</surname><given-names>V</given-names></name><name><surname>Reddy</surname><given-names>BAA</given-names></name><name><surname>Ruthrotha Selvi</surname><given-names>B</given-names></name><name><surname>Sukanya</surname><given-names>MS</given-names></name><name><surname>Kundu</surname><given-names>TK</given-names></name></person-group><year iso-8601-date="2007">2007</year><article-title>Small molecule modulators in epigenetics: Implications in gene expression and therapeutics</article-title><source>Sub-Cellular Biochemistry</source><volume>41</volume><fpage>397</fpage><lpage>428</lpage><pub-id pub-id-type="doi">10.1007/1-4020-5466-1_18</pub-id><pub-id pub-id-type="pmid">17484138</pub-id></element-citation></ref><ref id="bib67"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Taherian Fard</surname><given-names>A</given-names></name><name><surname>Ragan</surname><given-names>MA</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Quantitative modelling of the waddington epigenetic landscape</article-title><source>Methods in Molecular Biology</source><volume>1975</volume><fpage>157</fpage><lpage>171</lpage><pub-id pub-id-type="doi">10.1007/978-1-4939-9224-9_7</pub-id><pub-id pub-id-type="pmid">31062309</pub-id></element-citation></ref><ref id="bib68"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Thakore</surname><given-names>PI</given-names></name><name><surname>Black</surname><given-names>JB</given-names></name><name><surname>Hilton</surname><given-names>IB</given-names></name><name><surname>Gersbach</surname><given-names>CA</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Editing the epigenome: Technologies for programmable transcription and epigenetic modulation</article-title><source>Nature Methods</source><volume>13</volume><fpage>127</fpage><lpage>137</lpage><pub-id pub-id-type="doi">10.1038/nmeth.3733</pub-id><pub-id pub-id-type="pmid">26820547</pub-id></element-citation></ref><ref id="bib69"><element-citation publication-type="journal"><person-group person-group-type="author"><collab>The ENCODE Project Consortium</collab></person-group><year iso-8601-date="2012">2012</year><article-title>An integrated encyclopedia of DNA elements in the human genome</article-title><source>Nature</source><volume>489</volume><fpage>57</fpage><lpage>74</lpage><pub-id pub-id-type="doi">10.1038/nature11247</pub-id></element-citation></ref><ref id="bib70"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Virtanen</surname><given-names>P</given-names></name><name><surname>Gommers</surname><given-names>R</given-names></name><name><surname>Oliphant</surname><given-names>TE</given-names></name><name><surname>Haberland</surname><given-names>M</given-names></name><name><surname>Reddy</surname><given-names>T</given-names></name><name><surname>Cournapeau</surname><given-names>D</given-names></name><name><surname>Burovski</surname><given-names>E</given-names></name><name><surname>Peterson</surname><given-names>P</given-names></name><name><surname>Weckesser</surname><given-names>W</given-names></name><name><surname>Bright</surname><given-names>J</given-names></name><name><surname>van der Walt</surname><given-names>SJ</given-names></name><name><surname>Brett</surname><given-names>M</given-names></name><name><surname>Wilson</surname><given-names>J</given-names></name><name><surname>Millman</surname><given-names>KJ</given-names></name><name><surname>Mayorov</surname><given-names>N</given-names></name><name><surname>Nelson</surname><given-names>ARJ</given-names></name><name><surname>Jones</surname><given-names>E</given-names></name><name><surname>Kern</surname><given-names>R</given-names></name><name><surname>Larson</surname><given-names>E</given-names></name><name><surname>Carey</surname><given-names>CJ</given-names></name><name><surname>Polat</surname><given-names>İ</given-names></name><name><surname>Feng</surname><given-names>Y</given-names></name><name><surname>Moore</surname><given-names>EW</given-names></name><name><surname>VanderPlas</surname><given-names>J</given-names></name><name><surname>Laxalde</surname><given-names>D</given-names></name><name><surname>Perktold</surname><given-names>J</given-names></name><name><surname>Cimrman</surname><given-names>R</given-names></name><name><surname>Henriksen</surname><given-names>I</given-names></name><name><surname>Quintero</surname><given-names>EA</given-names></name><name><surname>Harris</surname><given-names>CR</given-names></name><name><surname>Archibald</surname><given-names>AM</given-names></name><name><surname>Ribeiro</surname><given-names>AH</given-names></name><name><surname>Pedregosa</surname><given-names>F</given-names></name><name><surname>van Mulbregt</surname><given-names>P</given-names></name><name><surname>Vijaykumar</surname><given-names>A</given-names></name><name><surname>Bardelli</surname><given-names>AP</given-names></name><name><surname>Rothberg</surname><given-names>A</given-names></name><name><surname>Hilboll</surname><given-names>A</given-names></name><name><surname>Kloeckner</surname><given-names>A</given-names></name><name><surname>Scopatz</surname><given-names>A</given-names></name><name><surname>Lee</surname><given-names>A</given-names></name><name><surname>Rokem</surname><given-names>A</given-names></name><name><surname>Woods</surname><given-names>CN</given-names></name><name><surname>Fulton</surname><given-names>C</given-names></name><name><surname>Masson</surname><given-names>C</given-names></name><name><surname>Häggström</surname><given-names>C</given-names></name><name><surname>Fitzgerald</surname><given-names>C</given-names></name><name><surname>Nicholson</surname><given-names>DA</given-names></name><name><surname>Hagen</surname><given-names>DR</given-names></name><name><surname>Pasechnik</surname><given-names>DV</given-names></name><name><surname>Olivetti</surname><given-names>E</given-names></name><name><surname>Martin</surname><given-names>E</given-names></name><name><surname>Wieser</surname><given-names>E</given-names></name><name><surname>Silva</surname><given-names>F</given-names></name><name><surname>Lenders</surname><given-names>F</given-names></name><name><surname>Wilhelm</surname><given-names>F</given-names></name><name><surname>Young</surname><given-names>G</given-names></name><name><surname>Price</surname><given-names>GA</given-names></name><name><surname>Ingold</surname><given-names>GL</given-names></name><name><surname>Allen</surname><given-names>GE</given-names></name><name><surname>Lee</surname><given-names>GR</given-names></name><name><surname>Audren</surname><given-names>H</given-names></name><name><surname>Probst</surname><given-names>I</given-names></name><name><surname>Dietrich</surname><given-names>JP</given-names></name><name><surname>Silterra</surname><given-names>J</given-names></name><name><surname>Webber</surname><given-names>JT</given-names></name><name><surname>Slavič</surname><given-names>J</given-names></name><name><surname>Nothman</surname><given-names>J</given-names></name><name><surname>Buchner</surname><given-names>J</given-names></name><name><surname>Kulick</surname><given-names>J</given-names></name><name><surname>Schönberger</surname><given-names>JL</given-names></name><name><surname>de Miranda Cardoso</surname><given-names>JV</given-names></name><name><surname>Reimer</surname><given-names>J</given-names></name><name><surname>Harrington</surname><given-names>J</given-names></name><name><surname>Rodríguez</surname><given-names>JLC</given-names></name><name><surname>Nunez-Iglesias</surname><given-names>J</given-names></name><name><surname>Kuczynski</surname><given-names>J</given-names></name><name><surname>Tritz</surname><given-names>K</given-names></name><name><surname>Thoma</surname><given-names>M</given-names></name><name><surname>Newville</surname><given-names>M</given-names></name><name><surname>Kümmerer</surname><given-names>M</given-names></name><name><surname>Bolingbroke</surname><given-names>M</given-names></name><name><surname>Tartre</surname><given-names>M</given-names></name><name><surname>Pak</surname><given-names>M</given-names></name><name><surname>Smith</surname><given-names>NJ</given-names></name><name><surname>Nowaczyk</surname><given-names>N</given-names></name><name><surname>Shebanov</surname><given-names>N</given-names></name><name><surname>Pavlyk</surname><given-names>O</given-names></name><name><surname>Brodtkorb</surname><given-names>PA</given-names></name><name><surname>Lee</surname><given-names>P</given-names></name><name><surname>McGibbon</surname><given-names>RT</given-names></name><name><surname>Feldbauer</surname><given-names>R</given-names></name><name><surname>Lewis</surname><given-names>S</given-names></name><name><surname>Tygier</surname><given-names>S</given-names></name><name><surname>Sievert</surname><given-names>S</given-names></name><name><surname>Vigna</surname><given-names>S</given-names></name><name><surname>Peterson</surname><given-names>S</given-names></name><name><surname>More</surname><given-names>S</given-names></name><name><surname>Pudlik</surname><given-names>T</given-names></name><name><surname>Oshima</surname><given-names>T</given-names></name><name><surname>Pingel</surname><given-names>TJ</given-names></name><name><surname>Robitaille</surname><given-names>TP</given-names></name><name><surname>Spura</surname><given-names>T</given-names></name><name><surname>Jones</surname><given-names>TR</given-names></name><name><surname>Cera</surname><given-names>T</given-names></name><name><surname>Leslie</surname><given-names>T</given-names></name><name><surname>Zito</surname><given-names>T</given-names></name><name><surname>Krauss</surname><given-names>T</given-names></name><name><surname>Upadhyay</surname><given-names>U</given-names></name><name><surname>Halchenko</surname><given-names>YO</given-names></name><name><surname>Vázquez-Baeza</surname><given-names>Y</given-names></name><collab>SciPy 1.0 Contributors</collab></person-group><year iso-8601-date="2020">2020</year><article-title>SciPy 1.0: Fundamental algorithms for scientific computing in Python</article-title><source>Nature Methods</source><volume>17</volume><fpage>261</fpage><lpage>272</lpage><pub-id pub-id-type="doi">10.1038/s41592-019-0686-2</pub-id><pub-id pub-id-type="pmid">32015543</pub-id></element-citation></ref><ref id="bib71"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname><given-names>K</given-names></name><name><surname>Escobar</surname><given-names>M</given-names></name><name><surname>Li</surname><given-names>J</given-names></name><name><surname>Mahata</surname><given-names>B</given-names></name><name><surname>Goell</surname><given-names>J</given-names></name><name><surname>Shah</surname><given-names>S</given-names></name><name><surname>Cluck</surname><given-names>M</given-names></name><name><surname>Hilton</surname><given-names>IB</given-names></name></person-group><year iso-8601-date="2022">2022</year><article-title>Systematic comparison of CRISPR-based transcriptional activators uncovers gene-regulatory features of enhancer-promoter interactions</article-title><source>Nucleic Acids Research</source><volume>50</volume><fpage>7842</fpage><lpage>7855</lpage><pub-id pub-id-type="doi">10.1093/nar/gkac582</pub-id><pub-id pub-id-type="pmid">35849129</pub-id></element-citation></ref><ref id="bib72"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Weinert</surname><given-names>BT</given-names></name><name><surname>Narita</surname><given-names>T</given-names></name><name><surname>Satpathy</surname><given-names>S</given-names></name><name><surname>Srinivasan</surname><given-names>B</given-names></name><name><surname>Hansen</surname><given-names>BK</given-names></name><name><surname>Schölz</surname><given-names>C</given-names></name><name><surname>Hamilton</surname><given-names>WB</given-names></name><name><surname>Zucconi</surname><given-names>BE</given-names></name><name><surname>Wang</surname><given-names>WW</given-names></name><name><surname>Liu</surname><given-names>WR</given-names></name><name><surname>Brickman</surname><given-names>JM</given-names></name><name><surname>Kesicki</surname><given-names>EA</given-names></name><name><surname>Lai</surname><given-names>A</given-names></name><name><surname>Bromberg</surname><given-names>KD</given-names></name><name><surname>Cole</surname><given-names>PA</given-names></name><name><surname>Choudhary</surname><given-names>C</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Time-resolved analysis reveals rapid dynamics and broad scope of the CBP/p300 acetylome</article-title><source>Cell</source><volume>174</volume><fpage>231</fpage><lpage>244</lpage><pub-id pub-id-type="doi">10.1016/j.cell.2018.04.033</pub-id><pub-id pub-id-type="pmid">29804834</pub-id></element-citation></ref><ref id="bib73"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Xiang</surname><given-names>G</given-names></name><name><surname>Keller</surname><given-names>CA</given-names></name><name><surname>Giardine</surname><given-names>B</given-names></name><name><surname>An</surname><given-names>L</given-names></name><name><surname>Li</surname><given-names>Q</given-names></name><name><surname>Zhang</surname><given-names>Y</given-names></name><name><surname>Hardison</surname><given-names>RC</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>S3norm: simultaneous normalization of sequencing depth and signal-to-noise ratio in epigenomic data</article-title><source>Nucleic Acids Research</source><volume>48</volume><elocation-id>e43</elocation-id><pub-id pub-id-type="doi">10.1093/nar/gkaa105</pub-id><pub-id pub-id-type="pmid">32086521</pub-id></element-citation></ref><ref id="bib74"><element-citation publication-type="preprint"><person-group person-group-type="author"><name><surname>Xu</surname><given-names>K</given-names></name><name><surname>Zhang</surname><given-names>M</given-names></name><name><surname>Li</surname><given-names>J</given-names></name><name><surname>Du</surname><given-names>SS</given-names></name><name><surname>Kawarabayashi</surname><given-names>KI</given-names></name><name><surname>Jegelka</surname><given-names>S</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>How neural networks extrapolate: from feedforward to graph neural networks</article-title><source>arXiv</source><pub-id pub-id-type="doi">10.48550/arXiv.2009.11848</pub-id></element-citation></ref><ref id="bib75"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Yoon</surname><given-names>S</given-names></name><name><surname>Eom</surname><given-names>GH</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>HDAC and HDAC inhibitor: from cancer to cardiovascular diseases</article-title><source>Chonnam Medical Journal</source><volume>52</volume><fpage>1</fpage><lpage>11</lpage><pub-id pub-id-type="doi">10.4068/cmj.2016.52.1.1</pub-id><pub-id pub-id-type="pmid">26865995</pub-id></element-citation></ref><ref id="bib76"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Zhang</surname><given-names>Y</given-names></name><name><surname>Reinberg</surname><given-names>D</given-names></name></person-group><year iso-8601-date="2001">2001</year><article-title>Transcription regulation by histone methylation: interplay between different covalent modifications of the core histone tails</article-title><source>Genes &amp; Development</source><volume>15</volume><fpage>2343</fpage><lpage>2360</lpage><pub-id pub-id-type="doi">10.1101/gad.927301</pub-id><pub-id pub-id-type="pmid">11562345</pub-id></element-citation></ref><ref id="bib77"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Zhao</surname><given-names>W</given-names></name><name><surname>Xu</surname><given-names>Y</given-names></name><name><surname>Wang</surname><given-names>Y</given-names></name><name><surname>Gao</surname><given-names>D</given-names></name><name><surname>King</surname><given-names>J</given-names></name><name><surname>Xu</surname><given-names>Y</given-names></name><name><surname>Liang</surname><given-names>FS</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>Investigating crosstalk between H3K27 acetylation and H3K4 trimethylation in CRISPR/dCas-based epigenome editing and gene activation</article-title><source>Scientific Reports</source><volume>11</volume><fpage>1</fpage><lpage>11</lpage><pub-id pub-id-type="doi">10.1038/s41598-021-95398-5</pub-id></element-citation></ref><ref id="bib78"><element-citation publication-type="preprint"><person-group person-group-type="author"><name><surname>Zheng</surname><given-names>X</given-names></name><name><surname>Cui</surname><given-names>J</given-names></name><name><surname>Wang</surname><given-names>Y</given-names></name><name><surname>Zhang</surname><given-names>J</given-names></name><name><surname>Wang</surname><given-names>C</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>CRSIPR-AI: a webtool for the efficacy prediction of CRISPR activation and interference</article-title><source>bioRxiv</source><pub-id pub-id-type="doi">10.1101/2021.12.02.470943</pub-id></element-citation></ref><ref id="bib79"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Zhong</surname><given-names>H</given-names></name><name><surname>Kim</surname><given-names>S</given-names></name><name><surname>Zhi</surname><given-names>D</given-names></name><name><surname>Cui</surname><given-names>X</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Predicting gene expression using DNA methylation in three human populations</article-title><source>PeerJ</source><volume>7</volume><elocation-id>e6757</elocation-id><pub-id pub-id-type="doi">10.7717/peerj.6757</pub-id><pub-id pub-id-type="pmid">31106051</pub-id></element-citation></ref><ref id="bib80"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Zhou</surname><given-names>X</given-names></name><name><surname>Blocker</surname><given-names>AW</given-names></name><name><surname>Airoldi</surname><given-names>EM</given-names></name><name><surname>O’Shea</surname><given-names>EK</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>A computational approach to map nucleosome positions and alternative chromatin states with base pair resolution</article-title><source>eLife</source><volume>5</volume><elocation-id>e16970</elocation-id><pub-id pub-id-type="doi">10.7554/eLife.16970</pub-id><pub-id pub-id-type="pmid">27623011</pub-id></element-citation></ref></ref-list><app-group><app id="appendix-1"><title>Appendix 1</title><sec sec-type="appendix" id="s8"><title>Experimental methods</title><sec sec-type="appendix" id="s8-1"><title>Identification and selection of gRNA</title><p>All gRNAs were designed following the same in silico identification algorithm. Genomic sequences were first identified using the UCSC Genome Browser and manipulated in Benchling. CRISPOR was used to separately identify all gRNA within a 500 bp window of each gene’s TSS. TSS coordinates were determined using Phantom prediction, gene annotation, and DNase Hypersensitivity data as visualized in the Genome Browser with the hg38 genome assembly. gRNA were selected from this list to optimize for coverage, predicted specificity (CFD score ≥ 80), and predicted on-target activity (Doench 16’ score) (<xref ref-type="bibr" rid="bib9">Doench et al., 2016</xref>). When no gRNA were available in a region, predicted score constraints were minimally relaxed to identify a gRNA.</p></sec><sec sec-type="appendix" id="s8-2"><title>Plasmid &amp; guide cloning, transfection, mRNA extraction, and qPCR</title><p>dCas9-p300 was cloned into a lentiviral plasmid backbone (Addgene# 83889) was a gift from Gersbach lab. gRNA were cloned using the molecular cloning pipeline described by the Zhang group. These gRNA were cloned into an isogenic minimal guide expression backbone utilizing the Gecko guide cloning strategy (<xref ref-type="bibr" rid="bib59">Shalem et al., 2014</xref>). This minimal guide cloning plasmid was a gift from Gersbach lab (Addgene#47108). Following sequence verification, gRNA tiling experiments were completed.</p></sec><sec sec-type="appendix" id="s8-3"><title>Cell culture</title><p>HEK293T (ATCC, CRL-11268) were purchased from ATCC and cultured using supplemented DMEM (10% FBS [Millipore], 1% penicillin/streptomycin [Gibco]). These cells were initially expanded and cryopreserved in 0.5% (vol/vol) DMSO containing supplemented DMEM at a concentration of 2E6/ml per vial. HEK293T cells were consistently passaged at 80% confluence using Trypsin/EDTA (Gibco) dissociation and passaged at a 1:10 ratio. Cells were disregarded after their 10th passage. No Mycoplasma contamination was observed.</p></sec><sec sec-type="appendix" id="s8-4"><title>gRNA tiling qPCR experiments</title><p>On day 0, healthy HEK293T cells (&lt;passage 10) were lifted with Trypsin EDTA, centrifuged, resuspended, and counted with a manual hemocytometer using Trypan blue to assess health. Cells with &gt;95% viability were seeded into 24-well plates with a consistent cell number per well (<inline-formula><alternatives><mml:math id="inf39"><mml:mn>1.5</mml:mn><mml:mo>×</mml:mo><mml:msup><mml:mn>10</mml:mn><mml:mrow class="MJX-TeXAtom-ORD"><mml:mn>5</mml:mn></mml:mrow></mml:msup></mml:math><tex-math id="inft39">\begin{document}$1.5\times 10^{5}$\end{document}</tex-math></alternatives></inline-formula>). 24 hours post plating, cells with confluence of between 70% and 90% with healthy phenotype were co-transfected with individual gRNA and dCas9-p300 plasmid DNA (mass = 500 ng, 125 ng, gRNA:375 ng dCas9-p300, using 1.5 ul Lipofectamine 3000) according to the manufacturer’s protocol. All qPCR experiments were conducted using 24 samples (two biological samples per condition) measured in the 96-well format. Additionally, these experiments individually test 11 uniquely targeting gRNA and utilized a non-targeting gRNA as a negative control for downstream analysis. Total cell mRNA was extracted using the QIAGEN RNeasy kit and protocol. Reverse transcription was then carried out using iScript Advanced reverse transcriptase (Bio-Rad) 750 ng of total RNA in a 10 ul reaction. From there, cDNA was diluted to 10 ng/ul based on the initial total RNA input. Then qPCR reactions were assembled in technical duplicate and consisted of the following: 45 ng (original mass) of reverse transcribed and diluted cDNA, Luna qPCR Mastermix (NEB), forward primer, and reverse primer. The appropriate primer set was used to target (a) the gene intended for transcriptional modulation and (b) GAPDH, a ubiquitously expressed gene used to normalize input cDNA mass.</p></sec><sec sec-type="appendix" id="s8-5"><title>MNase-seq</title><p>MNase sample processing was completed similarly to the previous methods (<xref ref-type="bibr" rid="bib8">Cui and Zhao, 2012</xref>) with modifications. HEK293T cells were grown in parallel for &gt;3 passages. Three biological replicates were processed together to minimize variance. Crosslinking was carried out on 20E7 HEK293T cells with 1% formaldehyde incubated for 10 minutes at 37°C prior to glycine quenching. Next, lysis and washing occurred, followed by nuclei isolation via 600 × <italic>g</italic> centrifugation. An initial optimization was performed on 2E6 purified nuclei using MNase amounts between 0.1 and 64 units of enzyme. RNase treatment as well as Proteinase K treatment and removal of crosslinks were performed as previously described (<xref ref-type="bibr" rid="bib43">McKnight et al., 2021</xref>). QIAquick PCR purification, sample DNA were visualized with the use of a 2% agarose gel and TapeStation (Agilent). The mononucleosomal band 150 bp was cut from the gel and purified using the QIAquick Gel purification. Heat was not used when melting gel to preserve AT rich regions. Following mono nucleosomal band purification, samples were quantified and size verified using a Tape Station (Agilent). Illumina libraries were produced using the NEBNext Ultra II DNA library preparation kit with NEBNext Dual Index Multiplex Oligos for Illumina using 1 µg of purified DNA as input and SPRI bead size selection after adapter ligation prior to index addition. Color-balanced unique i5 and i7 indices were used for each biological replicate to reduce confounds associated with index hopping. Prepared library concentrations and purity were determined on a tape station. Following verification, the three biological replicates were admixed with the same mass and sent to Azenta for sequencing on a single lane on the HISeq 3000/4000 platform with expected yield of 350 million paired end 150 bp length reads. This sequencing scheme was expected to yield a coverage of ∼10× for each biological replicate sample.</p></sec><sec sec-type="appendix" id="s8-6"><title>H3K27ac CUT&amp;RUN qPCR</title><p>H3K27ac CUT&amp;RUN qPCR CUT&amp;RUN was completed using the CUTANA ChIC/CUT&amp;RUN Kit by Epicypher (Catalog #: 14-1048). H3K27ac was bound using the Anti-Histone H3 (acetyl K27) antibody (Catalog#: ab4729) sold by Abcam. <italic>Escherichia coli</italic> spike-in DNA and Rabbit IgG (components of kit#: 14-1048) were used for qPCR input normalization and negative control, respectively. All experiments were performed in duplicate with three independent experiments. Briefly, p300 and individual gRNA were co-transfected into HEK293T cells, after 72 hours cells were detached, and CUT&amp;RUN was completed with identical cell number were used for each sample. qPCR was completed in technical duplicate using a primer set designed for amplification near the targeted promoter (<italic>CXCR4</italic> or <italic>TGFBR1</italic>). qPCR reactions were also completed using a previously described primer set for quantification of the <italic>E. coli</italic> gene uida. The ddT relative qPCR methods were used to analyze data, where uida Ct was used to normalize input and fold over rabbit IgG was calculated for each sample.</p></sec><sec sec-type="appendix" id="s8-7"><title>Perturb-seq scCRISPRa transduction</title><p>Data from Perturb-seq experiments were generated in a previous study (<xref ref-type="bibr" rid="bib20">Goell et al., 2024</xref>). Briefly, monoclonal K562 cell lines were generated by transducing cells (8 ug/mL) with respective constructs of interest across a range of %v/v and performing flow cytometry 2 days later. Wells receiving dilutions with 40% mCherry+ were selected and plated for monoclonal lines by limiting dilution. Monoclonal lines were transduced with varying titers to assess viral copy number. Library transduction with a 1.5% v/v was selected for the scCRISPRa experiment in which 500k cells were transduced with virus containing the gRNA library. Cells were spun out of polybrene-containing media and resuspended in standard K562 culture media. At 2 days post-transduction, 1 ug/mL puromycin was added to the culture. 9 days after transduction, cells were collected for scRNA-seq.</p></sec><sec sec-type="appendix" id="s8-8"><title>10X Genomics scRNA-sequencing with gRNA capture</title><p>Cells were harvested and prepared as per the 10X Genomics Single Cell Protocols Cell Preparation Guide. 10,000 cells were captured per lane using a 10X Chromium device. One lane was used for dCas9-p300 WT containing cells. Cells were captured using a 10X Chromium chip using the Chromium Next GEM Single Cell 3’ Reagents Kit v.3 with Feature Barcoding Technology for CRISPR screening. Final libraries were sequenced using a NovaSeq 6000 for each p300 screen. Gene expression and CRISPR Guide Capture transcript libraries were pooled at a 4:1 ratio for sequencing.</p></sec></sec><sec sec-type="appendix" id="s9"><title>Computational methods</title><sec sec-type="appendix" id="s9-1"><title>Analysis of Perturb-seq data</title><p>We analyzed single-cell RNA sequencing (scRNA-seq) data generated on the 10X Genomics platform as described previously (<xref ref-type="bibr" rid="bib20">Goell et al., 2024</xref>), corresponding to <inline-formula><alternatives><mml:math id="inf40"><mml:mi>G</mml:mi></mml:math><tex-math id="inft40">\begin{document}$G$\end{document}</tex-math></alternatives></inline-formula> genes and <inline-formula><alternatives><mml:math id="inf41"><mml:mi>r</mml:mi></mml:math><tex-math id="inft41">\begin{document}$r$\end{document}</tex-math></alternatives></inline-formula> guide RNAs (gRNAs) across <inline-formula><alternatives><mml:math id="inf42"><mml:mi>C</mml:mi></mml:math><tex-math id="inft42">\begin{document}$C$\end{document}</tex-math></alternatives></inline-formula> cells using the following steps:</p><sec sec-type="appendix" id="s9-1-1"><title>Quality control</title><list list-type="bullet" id="list1"><list-item><p>Cells were retained if the total gene count in a cell was between 1000 and 10,000, ensuring adequate complexity and excluding potential empty droplets or doublets.</p></list-item><list-item><p>Cells with mitochondrial gene content exceeding 10% were excluded to avoid including dying or stressed cells.</p></list-item></list></sec><sec sec-type="appendix" id="s9-1-2"><title>Normalization</title><p>To normalize the expression data, the following was applied to each gene’s raw count in each cell:<disp-formula id="equ9"><alternatives><mml:math id="m9"><mml:mrow><mml:mtext>Normalized Expression</mml:mtext><mml:mo>=</mml:mo><mml:mfrac><mml:mtext>Raw Count</mml:mtext><mml:mrow><mml:mtext>Total Counts per Cell</mml:mtext><mml:mo>+</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:mfrac></mml:mrow></mml:math><tex-math id="t9">\begin{document}$$\displaystyle \text{Normalized Expression}=\frac{\text{Raw Count}}{\text{Total Counts per Cell}+1}$$\end{document}</tex-math></alternatives></disp-formula></p></sec><sec sec-type="appendix" id="s9-1-3"><title>Computing Perturb-seq gene expression</title><p>For each gene <inline-formula><alternatives><mml:math id="inf43"><mml:mi>g</mml:mi></mml:math><tex-math id="inft43">\begin{document}$g$\end{document}</tex-math></alternatives></inline-formula>, targeted by a set of gRNAs denoted as <inline-formula><alternatives><mml:math id="inf44"><mml:msub><mml:mi>r</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi>g</mml:mi></mml:mrow></mml:msub></mml:math><tex-math id="inft44">\begin{document}$r_{g}$\end{document}</tex-math></alternatives></inline-formula>, we determined the impact of each specific gRNA <inline-formula><alternatives><mml:math id="inf45"><mml:msubsup><mml:mi>r</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi>g</mml:mi></mml:mrow><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi>i</mml:mi></mml:mrow></mml:msubsup></mml:math><tex-math id="inft45">\begin{document}$r_{g}^{i}$\end{document}</tex-math></alternatives></inline-formula> on its gene expression by performing the following steps:</p><list list-type="bullet" id="list2"><list-item><p><bold>Expression thresholding:</bold> Only cells with non-zero expression of gene <inline-formula><alternatives><mml:math id="inf46"><mml:mi>g</mml:mi></mml:math><tex-math id="inft46">\begin{document}$g$\end{document}</tex-math></alternatives></inline-formula> were selected for further analysis.</p></list-item><list-item><p><bold>gRNA-specific selection:</bold> From these cells, only those expressing the specific gRNA <inline-formula><alternatives><mml:math id="inf47"><mml:msubsup><mml:mi>r</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi>g</mml:mi></mml:mrow><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi>i</mml:mi></mml:mrow></mml:msubsup></mml:math><tex-math id="inft47">\begin{document}$r_{g}^{i}$\end{document}</tex-math></alternatives></inline-formula> and none other from <inline-formula><alternatives><mml:math id="inf48"><mml:msub><mml:mi>r</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi>g</mml:mi></mml:mrow></mml:msub></mml:math><tex-math id="inft48">\begin{document}$r_{g}$\end{document}</tex-math></alternatives></inline-formula> were retained.</p></list-item><list-item><p><bold>Pseudobulk quantification:</bold> For these cells, we computed the pseudobulk mean (<italic>μ</italic>) and standard deviation (<inline-formula><alternatives><mml:math id="inf49"><mml:mi>σ</mml:mi></mml:math><tex-math id="inft49">\begin{document}$\sigma$\end{document}</tex-math></alternatives></inline-formula>) of the expression levels of gene <inline-formula><alternatives><mml:math id="inf50"><mml:mi>g</mml:mi></mml:math><tex-math id="inft50">\begin{document}$g$\end{document}</tex-math></alternatives></inline-formula>.</p></list-item></list></sec><sec sec-type="appendix" id="s9-1-4"><title>Computing Perturb-seq fold-change</title><p>To establish the baseline expression of gene <inline-formula><alternatives><mml:math id="inf51"><mml:mi>g</mml:mi></mml:math><tex-math id="inft51">\begin{document}$g$\end{document}</tex-math></alternatives></inline-formula>, we considered cells not targeted by any gRNAs <inline-formula><alternatives><mml:math id="inf52"><mml:msubsup><mml:mi>r</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi>g</mml:mi></mml:mrow><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi>i</mml:mi></mml:mrow></mml:msubsup></mml:math><tex-math id="inft52">\begin{document}$r_{g}^{i}$\end{document}</tex-math></alternatives></inline-formula>. The pseudobulk mean (<inline-formula><alternatives><mml:math id="inf53"><mml:msub><mml:mi>μ</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mtext>control cells</mml:mtext></mml:mrow></mml:msub></mml:math><tex-math id="inft53">\begin{document}$\mu_{\text{control cells}}$\end{document}</tex-math></alternatives></inline-formula>) and standard deviation (<inline-formula><alternatives><mml:math id="inf54"><mml:msub><mml:mi>σ</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mtext>control cells</mml:mtext></mml:mrow></mml:msub></mml:math><tex-math id="inft54">\begin{document}$\sigma_{\text{control cells}}$\end{document}</tex-math></alternatives></inline-formula>) of gene <inline-formula><alternatives><mml:math id="inf55"><mml:mi>g</mml:mi></mml:math><tex-math id="inft55">\begin{document}$g$\end{document}</tex-math></alternatives></inline-formula>’s expression in these cells were calculated. The fold-change for gene <inline-formula><alternatives><mml:math id="inf56"><mml:mi>g</mml:mi></mml:math><tex-math id="inft56">\begin{document}$g$\end{document}</tex-math></alternatives></inline-formula> due to gRNA <inline-formula><alternatives><mml:math id="inf57"><mml:msubsup><mml:mi>r</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi>g</mml:mi></mml:mrow><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi>i</mml:mi></mml:mrow></mml:msubsup></mml:math><tex-math id="inft57">\begin{document}$r_{g}^{i}$\end{document}</tex-math></alternatives></inline-formula> was then quantified as<disp-formula id="equ10"><alternatives><mml:math id="m10"><mml:mrow><mml:mtext>fold-change</mml:mtext><mml:mo>=</mml:mo><mml:mfrac><mml:mi>μ</mml:mi><mml:msub><mml:mi>μ</mml:mi><mml:mtext>control cells</mml:mtext></mml:msub></mml:mfrac></mml:mrow></mml:math><tex-math id="t10">\begin{document}$$\displaystyle \text{fold-change}=\frac{\mu}{\mu_{\text{control cells}}}$$\end{document}</tex-math></alternatives></disp-formula></p></sec><sec sec-type="appendix" id="s9-1-5"><title>Plotting Perturb-seq fold-change against model predictions</title><p>In order to prepare <xref ref-type="fig" rid="fig6s4">Figure 6—figure supplement 4</xref>, we performed the following steps:</p><list list-type="bullet" id="list3"><list-item><p>Genes were included if there were at least two distinct 25 bp bins within 250 base pairs of the TSS with a gRNA targeting that gene and having a distinct expression fold-change.</p></list-item><list-item><p>gRNAs were included if the number of cells expressing <inline-formula><alternatives><mml:math id="inf58"><mml:msubsup><mml:mi>r</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi>g</mml:mi></mml:mrow><mml:mrow class="MJX-TeXAtom-ORD"><mml:mi>i</mml:mi></mml:mrow></mml:msubsup></mml:math><tex-math id="inft58">\begin{document}$r_{g}^{i}$\end{document}</tex-math></alternatives></inline-formula> was ≥ 2.</p></list-item><list-item><p>For each bin, the average Perturb-seq fold-change <italic>μ</italic> and the average predicted fold-change <inline-formula><alternatives><mml:math id="inf59"><mml:msub><mml:mi>μ</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mtext>predicted</mml:mtext></mml:mrow></mml:msub></mml:math><tex-math id="inft59">\begin{document}$\mu_{\text{predicted}}$\end{document}</tex-math></alternatives></inline-formula> were calculated as:<disp-formula id="equ11"><alternatives><mml:math id="m11"><mml:mrow><mml:msub><mml:mi>μ</mml:mi><mml:mtext>bin</mml:mtext></mml:msub><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mo largeop="true" symmetric="true">∑</mml:mo><mml:mtext>fold-change</mml:mtext></mml:mrow><mml:mtext>number of fold-change observations in the bin across gRNAs targeting the same 25 bp bin</mml:mtext></mml:mfrac></mml:mrow></mml:math><tex-math id="t11">\begin{document}$$\displaystyle  \mu_{\text{bin}}=\frac{\sum\text{fold-change}}{\text{number of fold-change observations in the bin across gRNAs targeting the same 25 bp bin}}$$\end{document}</tex-math></alternatives></disp-formula><disp-formula id="equ12"><alternatives><mml:math id="m12"><mml:mrow><mml:msub><mml:mi>μ</mml:mi><mml:mtext>predicted, bin</mml:mtext></mml:msub><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mo largeop="true" symmetric="true">∑</mml:mo><mml:mtext>predicted fold-change</mml:mtext></mml:mrow><mml:mtext>number of models used to make a prediction for the fold-change within this bin</mml:mtext></mml:mfrac></mml:mrow></mml:math><tex-math id="t12">\begin{document}$$\displaystyle  \mu_{\text{predicted, bin}}=\frac{\sum\text{predicted fold-change}}{\text{number of models used to make a prediction for the fold-change within this bin}}$$\end{document}</tex-math></alternatives></disp-formula></p></list-item><list-item><p>Ranks were assigned to <inline-formula><alternatives><mml:math id="inf60"><mml:msub><mml:mi>μ</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mtext>bin</mml:mtext></mml:mrow></mml:msub></mml:math><tex-math id="inft60">\begin{document}$\mu_{\text{bin}}$\end{document}</tex-math></alternatives></inline-formula> and <inline-formula><alternatives><mml:math id="inf61"><mml:msub><mml:mi>μ</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mtext>predicted, bin</mml:mtext></mml:mrow></mml:msub></mml:math><tex-math id="inft61">\begin{document}$\mu_{\text{predicted, bin}}$\end{document}</tex-math></alternatives></inline-formula> for comparison.</p></list-item><list-item><p>These ranks were then plotted against each other to evaluate the correlation between observed Perturb-seq fold-change and the model-predicted fold-change.</p></list-item></list><table-wrap id="app1table1" position="float"><label>Appendix 1—table 1.</label><caption><title>ChIP-seq <inline-formula><alternatives><mml:math id="inf62"><mml:mo>−</mml:mo><mml:msub><mml:mi>log</mml:mi><mml:mrow class="MJX-TeXAtom-ORD"><mml:mn>10</mml:mn></mml:mrow></mml:msub><mml:mo>⁡</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:mtext>p-values</mml:mtext><mml:mo stretchy="false">)</mml:mo></mml:math><tex-math id="inft62">\begin{document}$-\log_{10}(\text{p-values})$\end{document}</tex-math></alternatives></inline-formula> were obtained from the ENCODE Imputation Challenge where the ground truth data were available (corresponding to entries labeled T in the table).</title><p>Avocado imputations were downloaded from the ENCODE data portal, where ground truth data were not available (corresponding to entries labeled A in the table).</p></caption><table frame="hsides" rules="groups"><thead><tr><th align="left" valign="bottom">Cell type</th><th align="left" valign="bottom">polyA Plus RNA-seq</th><th align="left" valign="bottom">H3K36me3</th><th align="left" valign="bottom">H3K27me3</th><th align="left" valign="bottom">H3K27ac</th><th align="left" valign="bottom">H3K4me1</th><th align="left" valign="bottom">H3K4me3</th><th align="left" valign="bottom">H3K9me3</th></tr></thead><tbody><tr><td align="left" valign="bottom">IMR-90</td><td align="left" valign="bottom">T</td><td align="left" valign="bottom">T</td><td align="left" valign="bottom">T</td><td align="left" valign="bottom">T</td><td align="left" valign="bottom">T</td><td align="left" valign="bottom">T</td><td align="left" valign="bottom">T</td></tr><tr><td align="left" valign="bottom">H1-hESC</td><td align="left" valign="bottom">T</td><td align="left" valign="bottom">T</td><td align="left" valign="bottom">T</td><td align="left" valign="bottom">T</td><td align="left" valign="bottom">T</td><td align="left" valign="bottom">T</td><td align="left" valign="bottom">T</td></tr><tr><td align="left" valign="bottom">Trophoblast cell</td><td align="left" valign="bottom">T</td><td align="left" valign="bottom">T</td><td align="left" valign="bottom">T</td><td align="left" valign="bottom">T</td><td align="left" valign="bottom">T</td><td align="left" valign="bottom">T</td><td align="left" valign="bottom">T</td></tr><tr><td align="left" valign="bottom">Neural stem progenitor cell</td><td align="left" valign="bottom">T</td><td align="left" valign="bottom">T</td><td align="left" valign="bottom">T</td><td align="left" valign="bottom">T</td><td align="left" valign="bottom">T</td><td align="left" valign="bottom">T</td><td align="left" valign="bottom">T</td></tr><tr><td align="left" valign="bottom">K562</td><td align="left" valign="bottom">T</td><td align="left" valign="bottom">T</td><td align="left" valign="bottom">T</td><td align="left" valign="bottom">T</td><td align="left" valign="bottom">T</td><td align="left" valign="bottom">T</td><td align="left" valign="bottom">T</td></tr><tr><td align="left" valign="bottom">Heart left ventricle</td><td align="left" valign="bottom">T</td><td align="left" valign="bottom">T</td><td align="left" valign="bottom">T</td><td align="left" valign="bottom">T</td><td align="left" valign="bottom">T</td><td align="left" valign="bottom">T</td><td align="left" valign="bottom">T</td></tr><tr><td align="left" valign="bottom">Adrenal gland</td><td align="left" valign="bottom">T</td><td align="left" valign="bottom">T</td><td align="left" valign="bottom">T</td><td align="left" valign="bottom">T</td><td align="left" valign="bottom">T</td><td align="left" valign="bottom">T</td><td align="left" valign="bottom">T</td></tr><tr><td align="left" valign="bottom">Endocrine pancreas</td><td align="left" valign="bottom">T</td><td align="left" valign="bottom">T</td><td align="left" valign="bottom">T</td><td align="left" valign="bottom">T</td><td align="left" valign="bottom">T</td><td align="left" valign="bottom">T</td><td align="left" valign="bottom">T</td></tr><tr><td align="left" valign="bottom">Peripheral blood mononuclear cell</td><td align="left" valign="bottom">T</td><td align="left" valign="bottom">T</td><td align="left" valign="bottom">T</td><td align="left" valign="bottom">T</td><td align="left" valign="bottom">T</td><td align="left" valign="bottom">T</td><td align="left" valign="bottom">T</td></tr><tr><td align="left" valign="bottom">Amnion</td><td align="left" valign="bottom">T</td><td align="left" valign="bottom">T</td><td align="left" valign="bottom">T</td><td align="left" valign="bottom">T</td><td align="left" valign="bottom">T</td><td align="left" valign="bottom">T</td><td align="left" valign="bottom">T</td></tr><tr><td align="left" valign="bottom">Myoepithelial cell of mammary gland</td><td align="left" valign="bottom">T</td><td align="left" valign="bottom">T</td><td align="left" valign="bottom">T</td><td align="left" valign="bottom">A</td><td align="left" valign="bottom">T</td><td align="left" valign="bottom">T</td><td align="left" valign="bottom">T</td></tr><tr><td align="left" valign="bottom">Chorion</td><td align="left" valign="bottom">T</td><td align="left" valign="bottom">T</td><td align="left" valign="bottom">T</td><td align="left" valign="bottom">T</td><td align="left" valign="bottom">T</td><td align="left" valign="bottom">T</td><td align="left" valign="bottom">A</td></tr><tr><td align="left" valign="bottom">HEK293</td><td align="left" valign="bottom">T</td><td align="left" valign="bottom">T</td><td align="left" valign="bottom">A</td><td align="left" valign="bottom">T</td><td align="left" valign="bottom">T</td><td align="left" valign="bottom">T</td><td align="left" valign="bottom">T</td></tr></tbody></table></table-wrap><table-wrap id="app1table2" position="float"><label>Appendix 1—table 2.</label><caption><title>Endogenous gene expression of genes for which we generated dCas9-p300 epigenome editing data indicates that genes for which high fold-change was obtained are more likely to have low endogenous gene expression in HEK293T.</title><p>Cross-cell type Spearman provides a metric to assess how accurate our CNN model predictions are, on any given gene, across the 13 cell types.</p></caption><table frame="hsides" rules="groups"><thead><tr><th align="left" valign="bottom">Gene</th><th align="left" valign="bottom">HEK293 (SRR3997504) TPM</th><th align="left" valign="bottom">HEK293T (SRR13341848) TPM</th><th align="left" valign="bottom">HEK293T (SRR15013784) TPM</th><th align="left" valign="bottom">Maximum fold-change in dCas9-p300 data</th><th align="left" valign="bottom">Cross-cell type Spearman</th></tr></thead><tbody><tr><td align="left" valign="bottom">PRSS12</td><td align="left" valign="bottom">12.710</td><td align="left" valign="bottom">8.448</td><td align="left" valign="bottom">6.910</td><td align="left" valign="bottom">2.380</td><td align="left" valign="bottom">0.896</td></tr><tr><td align="left" valign="bottom">CXCR4</td><td align="left" valign="bottom">11.974</td><td align="left" valign="bottom">2.826</td><td align="left" valign="bottom">8.216</td><td align="left" valign="bottom">5.365</td><td align="left" valign="bottom">0.852</td></tr><tr><td align="left" valign="bottom">TGFBR1</td><td align="left" valign="bottom">0.725</td><td align="left" valign="bottom">3.254</td><td align="left" valign="bottom">8.029</td><td align="left" valign="bottom">3.675</td><td align="left" valign="bottom">0.689</td></tr><tr><td align="left" valign="bottom">C2CD4B</td><td align="left" valign="bottom">0.306</td><td align="left" valign="bottom">0.000</td><td align="left" valign="bottom">0.000</td><td align="left" valign="bottom">591.312</td><td align="left" valign="bottom">0.726</td></tr><tr><td align="left" valign="bottom">CD79A</td><td align="left" valign="bottom">0.280</td><td align="left" valign="bottom">0.207</td><td align="left" valign="bottom">0.127</td><td align="left" valign="bottom">127.094</td><td align="left" valign="bottom">0.364</td></tr><tr><td align="left" valign="bottom">SOX11</td><td align="left" valign="bottom">0.051</td><td align="left" valign="bottom">0.131</td><td align="left" valign="bottom">0.209</td><td align="left" valign="bottom">14.245</td><td align="left" valign="bottom">0.846</td></tr><tr><td align="left" valign="bottom">MYO1G</td><td align="left" valign="bottom">0.000</td><td align="left" valign="bottom">0.016</td><td align="left" valign="bottom">0.000</td><td align="left" valign="bottom">37.948</td><td align="left" valign="bottom">0.621</td></tr><tr><td align="left" valign="bottom">CYP17A1</td><td align="left" valign="bottom">0.000</td><td align="left" valign="bottom">0.000</td><td align="left" valign="bottom">0.000</td><td align="left" valign="bottom">6,549.110</td><td align="left" valign="bottom">0.397</td></tr></tbody></table></table-wrap></sec></sec></sec></app></app-group></back><sub-article article-type="editor-report" id="sa0"><front-stub><article-id pub-id-type="doi">10.7554/eLife.92991.4.sa0</article-id><title-group><article-title>eLife Assessment</article-title></title-group><contrib-group><contrib contrib-type="author"><name><surname>Khalil</surname><given-names>Ahmad S</given-names></name><role specific-use="editor">Reviewing Editor</role><aff><institution>Boston University</institution><country>United States</country></aff></contrib></contrib-group><kwd-group kwd-group-type="evidence-strength"><kwd>Solid</kwd></kwd-group><kwd-group kwd-group-type="claim-importance"><kwd>Useful</kwd></kwd-group></front-stub><body><p>This study presents an advance in efforts to use histone post-translational modification (PTM) data to model gene expression and predict epigenetic editing activity. Such models are broadly <bold>useful</bold> to the research community, especially ones that can model and predict epigenetic editing activity, which is novel; additionally, the authors have nicely integrated datasets across cell types into their model. The work is mostly <bold>solid</bold>, but it would be strengthened by performing further comparisons to existing methods that predict gene expression from PTM data and from more comprehensive functional validation of model-predicted epigenome editing outcomes beyond dCas9-p300 based perturbations. This work will be of interest to the epigenetics and computational modeling communities.</p></body></sub-article><sub-article article-type="referee-report" id="sa1"><front-stub><article-id pub-id-type="doi">10.7554/eLife.92991.4.sa1</article-id><title-group><article-title>Reviewer #1 (Public review):</article-title></title-group><contrib-group><contrib contrib-type="author"><anonymous/><role specific-use="referee">Reviewer</role></contrib></contrib-group></front-stub><body><p>Batra, Cabrera and Spence et al. present a model which integrates histone posttranslational modification (PTM) data across cell models to predict gene expression with the goal of using this model to better understand epigenetic editing. This gene expression prediction model approach is useful if (a) it predicts gene expression in specific cell lines (b) it predicts expression values rather than a rank or bin, (c) if it helps us to better understand the biology of gene expression or (d) it helps us to understand epigenome editing activity. Problematically for points (a) and (b) it is easier to directly measure gene expression than to measure multiple PTMs and so the real usefulness of this approach mostly relates to (c) and (d).</p><p>Other approaches have been published that use histone PTM to predict expression (e.g. PMID 27587684, 36588793). Is this model better in some way? No comparisons are made, although a claim is made that direct comparisons are difficult. I appreciate that the authors have not used the histone PTM data to predict gene expression levels of an &quot;average cell&quot; but rather that they are predicting expression within specific cell types or for unseen cell types. Approaches that predict expression levels are much more useful, whereas some previous approaches have only predicted expressed or not expressed or a rank order or bin-based ranking. The paper does not seem to have substantial novel insights into understanding the biology of gene expression.</p><p>The approach of using this model to predict epigenetic editor activity on transcription is interesting and to my knowledge novel although only examined in the context of a p300 editor. As the author point out the interpretation of the epigenetic editing data is convoluted by things like sgRNA activity scoring and to fully understand the results likely would require histone PTM profiling and maybe dCas9 ChIP-seq for each sgRNA which would be a substantial amount of work.</p><p>Furthermore from the model evaluation of H3K9me3 is seems the model is performing modestly for other forms of epigenetic or transcriptional editing- e.g. we know for the best studied transcriptional editor which is CRISPRi (dCas9-KRAB) that recruitment to a locus is associated with robust gene repression across the genome and is associated with H3K9me3 deposition by recruitment of KAP1/HP1/SETDB1 (PMID: 35688146, 31980609, 27980086, 26501517).</p><p>One concern overall with this approach is that dCas9-p300 has been observed to induce sgRNA independent off target H3K27Ac (<ext-link ext-link-type="uri" xlink:href="https://www.ncbi.nlm.nih.gov/pmc/articles/PMC8349887/">https://www.ncbi.nlm.nih.gov/pmc/articles/PMC8349887/</ext-link> see Figure S5D) which could convolute interpretation of this type of experiment for the model.</p><p>Comments on revisions: This resubmission adds a comparison to existing gene prediction methods, but add no new confirmation experiments with predicting epigenome editing efficiency and had only one minor text edit.</p></body></sub-article><sub-article article-type="referee-report" id="sa2"><front-stub><article-id pub-id-type="doi">10.7554/eLife.92991.4.sa2</article-id><title-group><article-title>Reviewer #2 (Public review):</article-title></title-group><contrib-group><contrib contrib-type="author"><anonymous/><role specific-use="referee">Reviewer</role></contrib></contrib-group></front-stub><body><p>Summary:</p><p>The authors build a gene expression model based on histone post-translational modifications, and find that H3K27ac is correlated with gene expression. They compare to other gene prediction methods such as DeepChrome. They proceed to perturb H3K27ac at 13 gene promoters in two cell types, and measure gene expression changes to test their model.</p><p>Strengths:</p><p>The combination of multiple methods to model expression, along with utilizing 6 histone datasets in 13 cell types allowed the authors to build a model that correlates between 0.7-0.79 with gene expression.</p><p>They compare three cells types to other prediction models, and this figure should be included in the main figures.</p><p>They use dCas9-p300 fusions to perturb H3K27ac and monitor gene expression to test their model. Ranked correlations of the HEK293 data showed some support for the predictions after perturbation of H3K27ac.</p><p>Weaknesses:</p><p>The authors state in the latest submission that the primary use case of this work is related to predicting epigenome editing outcomes, not predicting gene expression from chromatin. However the first four figures all relate to gene expression prediction. The only main figure that shows epigenome editing prediction is panel 6E. If this authors wish to highlight the use case of this work they should redo figures, including moving panels from current supplemental figures to show this.</p><p>The perturbation of 5 genes in K562 with perturb-seq data shows a modest correlation of ~0.5 and is still only shown in supplemental figures, which is odd as this is the true test case of their model in my opinion. The authors are then left to speculate the reasons why the outcome of epigenome editing doesn't fit their predictions, which highlights the limited value in the current version of this method.</p><p>As mentioned before, testing genes that were not expressed being most activated by dCas9-p300 weaken the correlations vs. looking at a broad range of different gene expression as the original model was trained on.</p><p>If the authors want this method to be used to predict outcomes of epigenome editing, expanding to dCas9-KRAB and other CRISPRa methods (SAM and VPR) would be useful. Those datasets are published and could be analyzed for this manuscript and show how the model holds up across cell types and epigenome editing methods.</p><p>The utility of this method as described here, to predict gRNA outcomes seems modest and limited. It is fairly trivial to test 10 or more gRNAs for a single gene to find the best one, and the authors show limited prediction and occasionally no benefit. For example, with CHD8 and CD79 the gRNA with the highest prediction had the lowest actual impact on gene expression of the gRNAs tested. For many other genes the gRNA's prediction and gene expression outcome show no correlation.</p></body></sub-article><sub-article article-type="author-comment" id="sa3"><front-stub><article-id pub-id-type="doi">10.7554/eLife.92991.4.sa3</article-id><title-group><article-title>Author response</article-title></title-group><contrib-group><contrib contrib-type="author"><name><surname>Batra</surname><given-names>Sanjit Singh</given-names></name><role specific-use="author">Author</role><aff><institution>University of California, Berkeley</institution><addr-line><named-content content-type="city">Berkeley</named-content></addr-line><country>United States</country></aff></contrib><contrib contrib-type="author"><name><surname>Cabrera</surname><given-names>Alan</given-names></name><role specific-use="author">Author</role><aff><institution>Rice University</institution><addr-line><named-content content-type="city">Houston</named-content></addr-line><country>United States</country></aff></contrib><contrib contrib-type="author"><name><surname>Spence</surname><given-names>Jeffrey P</given-names></name><role specific-use="author">Author</role><aff><institution>Stanford University</institution><addr-line><named-content content-type="city">Stanford</named-content></addr-line><country>United States</country></aff></contrib><contrib contrib-type="author"><name><surname>Goell</surname><given-names>Jacob</given-names></name><role specific-use="author">Author</role><aff><institution>Rice University</institution><addr-line><named-content content-type="city">Houston</named-content></addr-line><country>United States</country></aff></contrib><contrib contrib-type="author"><name><surname>Anand</surname><given-names>Selvalakshmi S</given-names></name><role specific-use="author">Author</role><aff><institution>Rice University</institution><addr-line><named-content content-type="city">Houston</named-content></addr-line><country>United States</country></aff></contrib><contrib contrib-type="author"><name><surname>Hilton</surname><given-names>Isaac</given-names></name><role specific-use="author">Author</role><aff><institution>Rice University</institution><addr-line><named-content content-type="city">Houston</named-content></addr-line><country>United States</country></aff></contrib><contrib contrib-type="author"><name><surname>Song</surname><given-names>Yun S</given-names></name><role specific-use="author">Author</role><aff><institution>University of California, Berkeley</institution><addr-line><named-content content-type="city">Berkeley</named-content></addr-line><country>United States</country></aff></contrib></contrib-group></front-stub><body><p>The following is the authors’ response to the previous reviews</p><disp-quote content-type="editor-comment"><p><bold>Reviewer #1 (Public Review):</bold></p><p>Batra, Cabrera and Spence et al. present a model which integrates histone posttranslational modification (PTM) data across cell models to predict gene expression with the goal of using this model to better understand epigenetic editing. This gene expression prediction model approach is useful if (a) it predicts gene expression in specific cell lines (b) it predicts expression values rather than a rank or bin, (c) if it helps us to better understand the biology of gene expression or (d) it helps us to understand epigenome editing activity. Problematically for point (a) and (b) it is easier to directly measure gene expression than to measure multiple PTMs and so the real usefulness of this approach mostly relates to (c) and (d).</p></disp-quote><p>We appreciate this point from Reviewer #1 and the instructive comments and helpful feedback on our study. We designed our approach keeping in mind that the primary use case is to understand how epigenome editing would affect gene expression.</p><disp-quote content-type="editor-comment"><p>Other approaches have been published that use histone PTM to predict expression (e.g. PMID 27587684, 36588793). Is this model better in some way? No comparisons are made although a claim is made that direct comparisons are difficult. I appreciate that the authors have not used the histone PTM data to predict gene expression levels of an &quot;average cell&quot; but rather that they are predicting expression within specific cell types or for unseen cell types. Approaches that predict expression levels are much more useful whereas some previous approaches have only predicted expressed or not expressed or a rank order or bin-based ranking. The paper does not seem to have substantial novel insights into understanding the biology of gene expression.</p></disp-quote><p>We thank Reviewer #1 again for this insightful comment. We have included citations for a series of papers (PMIDs: 27587684, 30147283, 36588793) that performed gene expression prediction using histone PTM data. However, each of these methods performs classification of gene expression as opposed to predicting the actual gene expression value via regression. Additionally, the referenced studies all work with Roadmap Epigenomics read-depth data as opposed to p-values obtained from the ENCODE pipelines, making it difficult to make direct comparisons. We outline in the Discussion section that by creating a comprehensive dataset of epigenome editing outcomes, which include quantification of histone PTMs before and after in situ 1 perturbations, will improve our understanding of the effects of dCas9-p300 on gene expression and assist in the design of gRNAs for achieving fine-tuned control over gene expression levels. In this revised version of our study, we have also added new data (Figure 3 – figure supplement 3) to further benchmark our model against others.</p><disp-quote content-type="editor-comment"><p>The approach of using this model to predict epigenetic editor activity on transcription is interesting and to my knowledge novel although only examined in the context of a p300 editor. As the author point out the interpretation of the epigenetic editing data is convoluted by things like sgRNA activity scoring and to fully understand the results likely would require histone PTM profiling and maybe dCas9 ChIP-seq for each sgRNA which would be a substantial amount of work.</p></disp-quote><p>We agree with the Reviewer and view these experiments as important components of future studies.</p><disp-quote content-type="editor-comment"><p>Furthermore from the model evaluation of H3K9me3 is seems the model is performing modestly for other forms of epigenetic or transcriptional editing- e.g. we know for the best studied transcriptional editor which is CRISPRi (dCas9-KRAB) that recruitment to a locus is associated with robust gene repression across the genome and is associated with H3K9me3 deposition by recruitment of KAP1/HP1/SETDB1 (PMID: 35688146, 31980609, 27980086, 26501517).</p></disp-quote><p>This is an interesting point. We have included new data (Figure 4 – figure supplement 1), that quantifies how sensitive the trained gene expression model is to perturbations in H3K9me3. Indeed our data suggests that the model predictions are sensitive to perturbations in H3K9me3. For instance, there is a clear decrease and a gradual increase as the position where the perturbation is performed moves from upstream to downstream of the TSS. Additionally, the magnitude of the predicted fold-change is a function of how much the H3K9me3 is perturbed and hence the magnitude of change would be even higher if the perturbation magnitude is increased. However, this precise magnitude is hard to estimate In the absence of experimental perturbation data for H3K9me3. Leveraging our model in combination with KRAB-based CRISPRi is an exciting and important aspect of future studies.</p><disp-quote content-type="editor-comment"><p>One concern overall with this approach is that dCas9-p300 has been observed to induce sgRNA independent off target H3K27Ac (<ext-link ext-link-type="uri" xlink:href="https://www.ncbi.nlm.nih.gov/pmc/articles/PMC8349887/">https://www.ncbi.nlm.nih.gov/pmc/articles/PMC8349887/</ext-link> see Figure S5D) which could convolute interpretation of this type of experiment for the model.</p></disp-quote><p>This remains an excellent point and indeed, we and others have observed that dCas9-p300 can result in off-target H3K27ac levels (both increased and suppressed) across the genome. Our study focused on p300, because the molecule is one of the few known proteins that can catalyze H3K27ac in the human genome, and H3K27ac remains a proxy for active genomic regulatory elements. Nevertheless, any off target activity of dCas9-p300 could certainly convolute our analyses. We have included language to address this caveat in our discussion.</p><disp-quote content-type="editor-comment"><p><bold>Reviewer #2 (Public review):</bold></p><p>Summary:</p><p>The authors build a gene expression model based on histone post-translational modifications, and find that H3K27ac is correlated with gene expression. They proceed to perturb H3K27ac at 13 gene promoters in two cell types, and measure gene expression changes to test their model.</p></disp-quote><p>We remain appreciative of the constructive feedback and input from Reviewer #2 on our manuscript.</p><disp-quote content-type="editor-comment"><p>Strengths:</p><p>The combination of multiple methods to model expression, along with utilizing 6 histone datasets in 13 cell types allowed the authors to build a model that correlates between 0.7-0.79 with gene expression. They use dCas9-p300 fusions to perturb H3K27ac and monitor gene expression to test their model. Ranked correlations of the HEK293 data showed some support for the predictions after perturbation of H3K27ac.</p><p>Weaknesses:</p><p>The perturbation of 5 genes in K562 with perturb-seq data shows a modest correlation of ~0.5 and isn't included in the main figures. The authors are then left to speculate reasons why the outcome of epigenome editing doesn't fit their predictions, which highlights the limited value in the current version of this method.</p></disp-quote><p>We agree with the reviewer’s suggestion and highlight in our conclusion that generating epigenome editing data across a variety of cell types and across many genes will help uncover the underlying mechanisms of gene expression modulation.</p><disp-quote content-type="editor-comment"><p>As mentioned before, testing genes that were not expressed being most activated by dCas9-p300 weaken the correlations vs. looking at a broad range of different gene expression as the original model was trained on.</p></disp-quote><p>We appreciate this comment from Reviewer #2. We note that the data generated from this dCas9-p300 perturb-seq experiment used gRNAs from a pre-existing library published previously (PMID: 37034704). While this library enabled deeper interrogation of dCas9-p300 driven effects compared to our previous revision, the gRNAs in this library were designed against genes associated with haploinsufficiency in neuronal cell types, and which were generally lowly-expressed in K562 cells. Further, we restricted our analysis here to promoter-proximal gRNAs (as opposed to enhancer-targeted gRNAs in the library), focusing our scope even more so. Thus the genes ultimately used for analysis are enriched for low expression.</p><disp-quote content-type="editor-comment"><p>If the authors want this method to be used to predict outcomes of epigenome editing, expanding to dCas9-KRAB and other CRISPRa methods (SAM and VPR) would be useful. Those datasets are published and could be analyzed for this manuscript.</p></disp-quote><p>This is an exciting suggestion from Reviewer #2. We agree, and view this as a component of future work in this area.</p><disp-quote content-type="editor-comment"><p>The authors don't compare their method to other prediction methods.</p></disp-quote><p>In this revised version of our study, we have also added new data (Figure 3 – figure supplement 3) to further benchmark our model against others. These data demonstrate that our CNN model outperforms existing approaches across multiple cell types.</p><disp-quote content-type="editor-comment"><p><bold>Recommendations for the authors:</bold></p><p><bold>Reviewer #2 (Recommendations for the authors):</bold></p><p>Looking at the individual genes in K562 shows a random looking range of predictions and observed, with the exception of Bcl11A which is one of two genes in this set of 5 that are not expressed. I will repeat my earlier comment, that epigenome editing and CRISPRa methods generally show the most upregulation with the lowest expressed genes. I speculate that plotting endogenous expression vs. outcome (assuming using all gRNAs within a reasonable and similar distance to TSS) would produce a correlation of -0.5 or greater and be as useful as this method.</p></disp-quote><p>We agree, and believe that this demonstrates more work is needed in this emerging research area.</p><disp-quote content-type="editor-comment"><p>The methods describe Perturb-seq analysis but not the bench experiments.</p></disp-quote><p>We have added the bench methods related to our Perturb-seq experiments to our revised manuscript under the Experimental Methods section in the Appendix.</p><disp-quote content-type="editor-comment"><p>I don't understand why the authors can't compare to other methods as that is fairly standard in new prediction papers. I get that others used REMC vs. ENCODE, and were rank or binary based, but the authors could use REMC data and/or convert their data to ranked or binary and still compare. Lacking that it's hard to judge this manuscript.</p></disp-quote><p>We have added benchmarking against existing methods as Figure 3 – figure supplement 3.</p></body></sub-article></article>