<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Archiving and Interchange DTD with MathML3 v1.3 20210610//EN"  "JATS-archivearticle1-3-mathml3.dtd"><article xmlns:ali="http://www.niso.org/schemas/ali/1.0/" xmlns:xlink="http://www.w3.org/1999/xlink" article-type="research-article" dtd-version="1.3"><front><journal-meta><journal-id journal-id-type="nlm-ta">elife</journal-id><journal-id journal-id-type="publisher-id">eLife</journal-id><journal-title-group><journal-title>eLife</journal-title></journal-title-group><issn publication-format="electronic" pub-type="epub">2050-084X</issn><publisher><publisher-name>eLife Sciences Publications, Ltd</publisher-name></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">95170</article-id><article-id pub-id-type="doi">10.7554/eLife.95170</article-id><article-id pub-id-type="doi" specific-use="version">10.7554/eLife.95170.3</article-id><article-version article-version-type="publication-state">version of record</article-version><article-categories><subj-group subj-group-type="display-channel"><subject>Research Article</subject></subj-group><subj-group subj-group-type="heading"><subject>Computational and Systems Biology</subject></subj-group><subj-group subj-group-type="heading"><subject>Genetics and Genomics</subject></subj-group></article-categories><title-group><article-title>Functional characteristics and computational model of abundant hyperactive loci in the human genome</article-title></title-group><contrib-group><contrib contrib-type="author" corresp="yes"><name><surname>Hudaiberdiev</surname><given-names>Sanjarbek</given-names></name><contrib-id authenticated="true" contrib-id-type="orcid">https://orcid.org/0000-0003-0860-5250</contrib-id><email>kyrgyzbala@gmail.com</email><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="other" rid="fund1"/><xref ref-type="fn" rid="con1"/><xref ref-type="fn" rid="conf1"/></contrib><contrib contrib-type="author" corresp="yes"><name><surname>Ovcharenko</surname><given-names>Ivan</given-names></name><email>ovcharen@nih.gov</email><xref ref-type="aff" rid="aff1">1</xref><xref ref-type="other" rid="fund1"/><xref ref-type="fn" rid="con2"/><xref ref-type="fn" rid="conf1"/></contrib><aff id="aff1"><label>1</label><institution-wrap><institution-id institution-id-type="ror">https://ror.org/01cwqze88</institution-id><institution>National Institute for Biotechnology and Information, National Library of Medicine, National Institutes of Health</institution></institution-wrap><addr-line><named-content content-type="city">Bethesda</named-content></addr-line><country>United States</country></aff></contrib-group><contrib-group content-type="section"><contrib contrib-type="editor"><name><surname>Altemose</surname><given-names>Nicolas</given-names></name><role>Reviewing Editor</role><aff><institution-wrap><institution-id institution-id-type="ror">https://ror.org/00f54p054</institution-id><institution>Stanford University</institution></institution-wrap><country>United States</country></aff></contrib><contrib contrib-type="senior_editor"><name><surname>Araújo</surname><given-names>Sofia J</given-names></name><role>Senior Editor</role><aff><institution-wrap><institution-id institution-id-type="ror">https://ror.org/021018s57</institution-id><institution>University of Barcelona</institution></institution-wrap><country>Spain</country></aff></contrib></contrib-group><pub-date publication-format="electronic" date-type="publication"><day>13</day><month>11</month><year>2024</year></pub-date><volume>13</volume><elocation-id>RP95170</elocation-id><history><date date-type="sent-for-review" iso-8601-date="2024-02-15"><day>15</day><month>02</month><year>2024</year></date></history><pub-history><event><event-desc>This manuscript was published as a preprint.</event-desc><date date-type="preprint" iso-8601-date="2023-12-15"><day>15</day><month>12</month><year>2023</year></date><self-uri content-type="preprint" xlink:href="https://doi.org/10.1101/2023.02.05.527203"/></event><event><event-desc>This manuscript was published as a reviewed preprint.</event-desc><date date-type="reviewed-preprint" iso-8601-date="2024-05-09"><day>09</day><month>05</month><year>2024</year></date><self-uri content-type="reviewed-preprint" xlink:href="https://doi.org/10.7554/eLife.95170.1"/></event><event><event-desc>The reviewed preprint was revised.</event-desc><date date-type="reviewed-preprint" iso-8601-date="2024-10-08"><day>08</day><month>10</month><year>2024</year></date><self-uri content-type="reviewed-preprint" xlink:href="https://doi.org/10.7554/eLife.95170.2"/></event></pub-history><permissions><ali:free_to_read/><license xlink:href="http://creativecommons.org/publicdomain/zero/1.0/"><ali:license_ref>http://creativecommons.org/publicdomain/zero/1.0/</ali:license_ref><license-p>This is an open-access article, free of all copyright, and may be freely reproduced, distributed, transmitted, modified, built upon, or otherwise used by anyone for any lawful purpose. The work is made available under the <ext-link ext-link-type="uri" xlink:href="http://creativecommons.org/publicdomain/zero/1.0/">Creative Commons CC0 public domain dedication</ext-link>.</license-p></license></permissions><self-uri content-type="pdf" xlink:href="elife-95170-v1.pdf"/><self-uri content-type="figures-pdf" xlink:href="elife-95170-figures-v1.pdf"/><abstract><p>Enhancers and promoters are classically considered to be bound by a small set of transcription factors (TFs) in a sequence-specific manner. This assumption has come under increasing skepticism as the datasets of ChIP-seq assays of TFs have expanded. In particular, high-occupancy target (HOT) loci attract hundreds of TFs with often no detectable correlation between ChIP-seq peaks and DNA-binding motif presence. Here, we used a set of 1003 TF ChIP-seq datasets (HepG2, K562, H1) to analyze the patterns of ChIP-seq peak co-occurrence in combination with functional genomics datasets. We identified 43,891 HOT loci forming at the promoter (53%) and enhancer (47%) regions. HOT promoters regulate housekeeping genes, whereas HOT enhancers are involved in tissue-specific process regulation. HOT loci form the foundation of human super-enhancers and evolve under strong negative selection, with some of these loci being located in ultraconserved regions. Sequence-based classification analysis of HOT loci suggested that their formation is driven by the sequence features, and the density of mapped ChIP-seq peaks across TF-bound loci correlates with sequence features and the expression level of flanking genes. Based on the affinities to bind to promoters and enhancers we detected five distinct clusters of TFs that form the core of the HOT loci. We report an abundance of HOT loci in the human genome and a commitment of 51% of all TF ChIP-seq binding events to HOT locus formation thus challenging the classical model of enhancer activity and propose a model of HOT locus formation based on the existence of large transcriptional condensates.</p></abstract><kwd-group kwd-group-type="author-keywords"><kwd>gene expression</kwd><kwd>high-occupancy target</kwd><kwd>HOT</kwd><kwd>transcriptional condensates</kwd></kwd-group><kwd-group kwd-group-type="research-organism"><title>Research organism</title><kwd>Human</kwd></kwd-group><funding-group><award-group id="fund1"><funding-source><institution-wrap><institution-id institution-id-type="FundRef">http://dx.doi.org/10.13039/100000002</institution-id><institution>National Institutes of Health</institution></institution-wrap></funding-source><principal-award-recipient><name><surname>Hudaiberdiev</surname><given-names>Sanjarbek</given-names></name><name><surname>Ovcharenko</surname><given-names>Ivan</given-names></name></principal-award-recipient></award-group><funding-statement>The funders had no role in study design, data collection and interpretation, or the decision to submit the work for publication.</funding-statement></funding-group><custom-meta-group><custom-meta specific-use="meta-only"><meta-name>Author impact statement</meta-name><meta-value>Large-scale multi-omics analysis of genomic high-occupancy target regions in humans suggests involvement of large transcriptional condensates.</meta-value></custom-meta><custom-meta specific-use="meta-only"><meta-name>publishing-route</meta-name><meta-value>prc</meta-value></custom-meta></custom-meta-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><p>Tissue -specificity of gene expression is orchestrated by the combination of transcription factors (TFs) that bind to regulatory regions such as promoters, enhancers, and silencers (<xref ref-type="bibr" rid="bib39">Moore et al., 2020</xref>; <xref ref-type="bibr" rid="bib20">Gorkin et al., 2020</xref>). Classically, an enhancer is thought to be bound by a few TFs that recognize a specific DNA motif at their cognate TF binding site (TFBS) through its DNA-binding domain and recruit other molecules necessary for catalyzing the transcriptional machinery (<xref ref-type="bibr" rid="bib17">Forsberg and Westin, 1991</xref>; <xref ref-type="bibr" rid="bib54">Serfling et al., 1985</xref>; <xref ref-type="bibr" rid="bib55">Sethi et al., 2020</xref>). Based on the arrangements of the TFBSs, also called ‘motif grammar’, the architecture of enhancers is commonly categorized into ‘enhanceosome’ and ‘billboard’ models (<xref ref-type="bibr" rid="bib58">Spitz and Furlong, 2012</xref>; <xref ref-type="bibr" rid="bib35">Long et al., 2016</xref>). In the enhanceosome model, a rigid grammar of motifs facilitates the formation of a single structure comprising multiple TFs which then activates the target gene (<xref ref-type="bibr" rid="bib61">Thanos and Maniatis, 1995</xref>; <xref ref-type="bibr" rid="bib36">Merika and Thanos, 2001</xref>). This model requires the presence of all the participating proteins. Under the billboard model, on the other hand, the TFBSs are independent of each other and function in an additive manner (<xref ref-type="bibr" rid="bib3">Arnosti and Kulkarni, 2005</xref>). However, as the catalogs of TF ChIP-seq assays have expanded thanks to the major collaborative projects such as ENCODE (<xref ref-type="bibr" rid="bib13">Davis et al., 2018</xref>) and modENCODE (<xref ref-type="bibr" rid="bib51">Roy et al., 2010</xref>), this assertion that the TFs interact with DNA through the strictly defined binding motifs has fallen under increasing contradiction with empirically observed patterns of DNA-binding regions of TFs. In particular, there have been reported genomic regions that seemingly get bound by a large number of TFs with no apparent DNA sequence specificity in terms of detectable binding motifs of corresponding motifs. These genomic loci have been dubbed high-occupancy target (HOT) regions and were detected in multiple species (<xref ref-type="bibr" rid="bib51">Roy et al., 2010</xref>; <xref ref-type="bibr" rid="bib40">Moorman et al., 2006</xref>; <xref ref-type="bibr" rid="bib19">Gerstein et al., 2010</xref>; <xref ref-type="bibr" rid="bib26">Kvon et al., 2012</xref>; <xref ref-type="bibr" rid="bib75">Yip et al., 2012</xref>).</p><p>Initially, these regions have been partially attributed to technical and statistical artifacts of the ChIP-seq protocol, resulting in a small list of blacklisted regions that are mostly located in unstructured DNA regions such as repetitive elements and low complexity regions (<xref ref-type="bibr" rid="bib60">Teytelman et al., 2013</xref>; <xref ref-type="bibr" rid="bib71">Wreczycka et al., 2019</xref>). These blacklisted regions have been later excluded from the analyses and they represent a small fraction of the mapped ChIP-seq peaks. In addition, various studies have proposed the idea that some DNA elements can serve as permissive TF binding platforms such as GC-rich promoters, CpG islands, R-loops, and G-quadruplexes (<xref ref-type="bibr" rid="bib60">Teytelman et al., 2013</xref>; <xref ref-type="bibr" rid="bib71">Wreczycka et al., 2019</xref>). Other studies have concluded that these regions are highly functionally consequential regions enriched in epigenetic signals of active regulatory elements such as histone modification regions and high chromatin accessibility (<xref ref-type="bibr" rid="bib51">Roy et al., 2010</xref>; <xref ref-type="bibr" rid="bib48">Ramaker et al., 2020</xref>; <xref ref-type="bibr" rid="bib45">Partridge et al., 2020</xref>).</p><p>Early studies of the subject have been limited in scope due to the small number of available TF ChIP-seq assays. There have been numerous studies in recent years with additional TFs across multiple cell lines. For instance, (<xref ref-type="bibr" rid="bib45">Partridge et al., 2020</xref>), studied the HOT loci in the context of 208 proteins including TFs, cofactors, and chromatin regulators which they called chromatin-associated proteins. They observed that the composition of the chromatin-associated proteins differs depending on whether the HOT locus is located in an enhancer or promoter. <xref ref-type="bibr" rid="bib71">Wreczycka et al., 2019</xref>, performed a cross-species analysis of HOT loci in the promoters of highly expressed genes, and established that some of the HOT loci correspond to the ‘hyper-ChIPable’ regions. <xref ref-type="bibr" rid="bib48">Ramaker et al., 2020</xref>, conducted a comparative study of HOT regions in multiple cell lines and detected putative driver motifs at the core segments of the HOT loci.</p><p>In this study, we used the most up-to-date set of TF ChIP-seq assays available from the ENCODE Project (<ext-link ext-link-type="uri" xlink:href="https://encodeproject.org/">https://encodeproject.org/</ext-link>) and incorporated functional genomics datasets such as 3D chromatin data (Hi-C), eQTLs, GWAS, and clinical disease variants to characterize and analyze the functional implications of the HOT loci. We report that the HOT loci are one of the prevalent modes of regulatory TF-DNA interactions; they represent active regulatory regions with distinct patterns of bound TFs manifested as clusters of promoter-specific, enhancer-specific, and chromatin-associated proteins. They are active during the embryonic stage and are enriched in disease-associated variants. Finally, we propose a model for the HOT regions based on the idea of the existence of large transcriptional condensates.</p></sec><sec id="s2" sec-type="results"><title>Results</title><sec id="s2-1"><title>HOT loci are one of the prevalent modes of TF-DNA interactions</title><p>To define and analyze the HOT loci, we used the most up-to-date catalog of ChIP-seq datasets (n=1003) of TFs obtained from the ENCODE Project assayed in HepG2, K562, and H1-hESC (H1) cells (545, 411, and 47 ChIP-seq assays, respectively, see Methods for details). While the TFs are defined as sequence-specific DNA-binding proteins that control the transcription of genes, the currently available ChIP-seq datasets include the assays of many other types of transcription-related proteins such as cofactors, coactivators, histone acetyltransferases, as well as RNA Polymerase 2 variants. Therefore, we collectively call all of these proteins DNA-associated proteins (DAPs). Using the datasets of DAPs, we overlaid all of the ChIP-seq peaks and obtained the densities of DAP binding sites across the human genome using a non-overlapping sliding window of length 400 bp and considered a binding site to be present in a given window if 8 bp centered at the summit of a ChIP-seq peak as overlapping. Given that the analyzed three cell lines contain varying numbers of assayed DAPs, we binned the loci according to the number of overlapping DAPs in a logarithmic scale with 10 intervals and defined HOT loci as those that fall to the highest four bins, which translates to those which contain on average &gt;18% of available DAPs for a given cell line (see Methods for a detailed description and justifications). This resulted in 25,928, 15,231, and 2732 HOT loci in HepG2, K562, and H1 cells, respectively. We applied our definition to the Roadmap Epigenomic ChIP-seq datasets and observed that the number of available ChIP-seq datasets significantly affects the resulting HOT loci. However, the HOT loci defined using the Roadmap Epigenomic datasets were almost entirely composed of subsets of the ENCODE-based HOT loci, comprising 50%, 62%, and 15% in HepG2, K562, and H1, respectively (<xref ref-type="supplementary-material" rid="supp1">Supplementary file 1, table S5</xref>). Importantly, we note that the distribution of the number of loci is not multimodal, but rather follows a uniform spectrum, and thus, this definition of HOT loci is ad hoc (<xref ref-type="fig" rid="fig1">Figure 1A</xref>, <xref ref-type="fig" rid="fig1s1">Figure 1—figure supplement 1</xref>). Therefore, in addition to the dichotomous classification of HOT and non-HOT loci, we use all of the DAP-bound loci to extract the correlations with studied metrics with the number of bound DAPs when necessary. Throughout the study, we used the loci from the HepG2 cell line as the primary dataset for analyses and used the K562 and H1 datasets when the comparative analysis was necessary.</p><fig-group><fig id="fig1" position="float"><label>Figure 1.</label><caption><title>High-occupancy target (HOT) loci are prevalent in the genome.</title><p>(<bold>A</bold>) Distribution of the number of loci by the number of overlapping peaks 400 bp loci. Loci are binned on a logarithmic scale (<xref ref-type="table" rid="table1">Table 1</xref>, Methods). The shaded region represents the HOT loci. (<bold>B</bold>) Prevalence of DNA-associated proteins (DAPs) in HOT loci. Each dot represents a DAP. X-axis: percentage of HOT loci in which DAP is present (e.g. MAX is present in 80% of HOT loci). Y-axis: percentage of total peaks of DAPs that are located in HOT loci (e.g. 45% of all the ChIP-seq peaks of MAX is located in the HOT loci). Dot color and size are proportional to the total number of ChIP-seq peaks of DAP. (<bold>C</bold>) Breakdown of HepG2 HOT loci to the promoter, intronic, and intergenic regions. (<bold>D</bold>) Fractions of HOT enhancer and promoter loci located in ATAC-seq. (<bold>E</bold>) Overlaps between the HOT enhancer, HOT promoter, super-enhancer, regular enhancer, H3K27ac, and H4K4me1 regions. Horizontal bars on bottom left represent the total number of loci of the corresponding class of loci. All of the visualized data is generated from the HepG2 cell line.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-95170-fig1-v1.tif"/></fig><fig id="fig1s1" position="float" specific-use="child-fig"><label>Figure 1—figure supplement 1.</label><caption><title>Distribution of the number of loci by the number of overlapping peaks 400 bp loci in K562 and H1.</title><p>Loci are binned on a logarithmic scale (<xref ref-type="table" rid="table1">Table 1</xref>, see Methods).</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-95170-fig1-figsupp1-v1.tif"/></fig><fig id="fig1s2" position="float" specific-use="child-fig"><label>Figure 1—figure supplement 2.</label><caption><title>Percentages of overlapping promoter (top row) and enhancer (bottom row) loci binned by bound DNA-associated proteins (DAPs) with histone modification regions in HepG2 (left column) and K562 (right column).</title></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-95170-fig1-figsupp2-v1.tif"/></fig><fig id="fig1s3" position="float" specific-use="child-fig"><label>Figure 1—figure supplement 3.</label><caption><title>Composition of the high-occupancy target (HOT) loci to promoter and enhancer regions based on the definitions used in this study, chromHMM states and ENCODE SCREEN annotations.</title></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-95170-fig1-figsupp3-v1.tif"/></fig><fig id="fig1s4" position="float" specific-use="child-fig"><label>Figure 1—figure supplement 4.</label><caption><title>Overlaps between the high-occupancy target (HOT) loci as reported in this study (<xref ref-type="bibr" rid="bib48">Ramaker et al., 2020</xref> and <xref ref-type="bibr" rid="bib8">Boyle et al., 2014</xref>).</title><p>Overlaps are calculated in terms of fractions of overlapping bps.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-95170-fig1-figsupp4-v1.tif"/></fig><fig id="fig1s5" position="float" specific-use="child-fig"><label>Figure 1—figure supplement 5.</label><caption><title>phastCons conservation scores of high-occupancy target (HOT) loci defined by this study (<xref ref-type="bibr" rid="bib48">Ramaker et al., 2020</xref> and <xref ref-type="bibr" rid="bib8">Boyle et al., 2014</xref>).</title><p>Bar plots depict median values, error bars are 95% confidence intervals. p-Values are Mann-Whitney U test results.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-95170-fig1-figsupp5-v1.tif"/></fig><fig id="fig1s6" position="float" specific-use="child-fig"><label>Figure 1—figure supplement 6.</label><caption><title>Compositions of high-occupancy target (HOT) loci as reported in this study (<xref ref-type="bibr" rid="bib48">Ramaker et al., 2020</xref> and <xref ref-type="bibr" rid="bib8">Boyle et al., 2014</xref>) in terms of promoter, intronic, and intergenic regions.</title></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-95170-fig1-figsupp6-v1.tif"/></fig></fig-group><p>Although the HOT loci represent only 5% of all the DAP-bound loci in HepG2, they contain 51% of all mapped ChIP-seq peaks. The fraction of the ChIP-seq peaks of each DAP overlapping HOT loci varies from 0% to 91%, with an average of 65% (<xref ref-type="fig" rid="fig1">Figure 1B</xref>, y-axis). Among the DAPs that are present in the highest fraction of HOT loci are (<xref ref-type="fig" rid="fig1">Figure 1B</xref>, x-axis) SAP130, MAX, ARID4B, ZGPAT, HDAC1, MED1, TFAP4, and SOX6. The abundance of histone deacetylase-related factors mixed with transcriptional activators suggests that the regulatory functions of HOT loci are a complex interplay of activation and repression. RNA Polymerase 2 (POLR2) is present in 42% of HOT loci arguing for active transcription at or in the proximity of HOT loci (including mRNA and eRNA transcription). When the fraction of peaks of individual DAPs overlapping with the HOT loci are considered (<xref ref-type="fig" rid="fig1">Figure 1B</xref>, y-axis), DAPs with &gt;90% overlap are GMEB2 (essential for replication of parvoviruses), ZHX3 (zinc finger transcriptional repressor), and YEATS2 (subunit of acetyltransferase complex). Whereas the DAPs that are least associated with HOT loci (&lt;5%) are ZNF282 (transcriptional repressor), MAFK, EZH2 (histone methyltransferase), and TRIM22 (ubiquitin ligase). The fact that HOT loci harbor more than half of the ChIP-seq peaks suggests that the HOT loci are one of the prevalent modes of TF-DNA interactions rather than an exceptional case, as has been initially suggested by earlier studies (<xref ref-type="bibr" rid="bib60">Teytelman et al., 2013</xref>; <xref ref-type="bibr" rid="bib71">Wreczycka et al., 2019</xref>).</p><p>Around half of the HOT loci (51%) are located in promoter regions (46% in primary promoters and 5% in alternative promoters), 25% in intronic regions, and only 24% are in intergenic regions with 9% being located &gt;50 kb away from promoters, suggesting that the HOT loci are mainly clustered in vicinities (promoters and introns) of transcription start sites and therefore potentially playing essential roles in the regulation of nearby genes (<xref ref-type="fig" rid="fig1">Figure 1C</xref>). When considering the non-promoter HOT loci, we observed that they were universally located in regions of H3K27ac or H3K4me1, indicating that they are active enhancers (<xref ref-type="fig" rid="fig1s2">Figure 1—figure supplement 2</xref>). When comparing the definitions of promoters and enhancers based on chromHMM states and ENCODE SCREEN annotations, the composition of HOT loci in relation to promoters and enhancers showed similar fractions (<xref ref-type="fig" rid="fig1s3">Figure 1—figure supplement 3</xref>). Both HOT promoters and enhancers are almost entirely located in the chromatin-accessible regions (97% and 93% of the total sequence lengths, respectively, <xref ref-type="fig" rid="fig1">Figure 1D</xref>). We compared our definition of the HOT loci to those reported in <xref ref-type="bibr" rid="bib48">Ramaker et al., 2020</xref>, and <xref ref-type="bibr" rid="bib8">Boyle et al., 2014</xref>. We observed that because these two studies define HOT loci using 2 kb windows, they cover a larger fraction of the genome. Our set of HOT loci largely consisted of subsets of those defined in these two studies, with overlap percentages of 81%, 93%, and 100% in HepG2, K562, and H1, respectively (<xref ref-type="fig" rid="fig1s4">Figure 1—figure supplement 4</xref>). Further analysis revealed that our set of HOT loci primarily constitutes the ‘core’ and more conserved (<xref ref-type="fig" rid="fig1s5">Figure 1—figure supplement 5</xref>) regions of HOT loci defined in the mentioned studies, while their composition in terms of promoter, intronic, and intergenic regions is similar (<xref ref-type="fig" rid="fig1s6">Figure 1—figure supplement 6</xref>), suggesting that the three definitions point to loci with similar characteristics.</p><p>To further dissect the composition of HOT enhancer loci, we compared them to super-enhancers as defined in the study by <xref ref-type="bibr" rid="bib70">Whyte et al., 2013</xref>, and a set of regular enhancers (Methods). Overall, 31% of HOT enhancers and 16% of HOT promoters are located in super-enhancers, while 97% of all HOT loci overlap with H3K27ac or H3K4me1 regions (<xref ref-type="fig" rid="fig1">Figure 1E</xref>). While HOT enhancers and promoters appear to provide a critical foundation for super-enhancer formation, they represent only a small fraction of super-enhancer sequences overall accounting for 9% of combined super-enhancer length.</p><p>A 400 bp HOT locus, on average, harbors 125 DAP peaks in HepG2. However, the peaks of DAPs are not uniformly distributed across HOT loci. There are 68 DAPs with &gt;80% of all of the peaks located in HOT loci (<xref ref-type="fig" rid="fig1">Figure 1B</xref>). To analyze the signatures of unique DAPs in HOT loci, we performed a PCA where each HOT locus is represented by a binary (presence/absence) vector of length equal to the total number of DAPs analyzed. This analysis showed that the principal component 1 (PC1) is correlated with the total number of distinct DAPs located at a given HOT locus (<xref ref-type="fig" rid="fig2s1">Figure 2—figure supplement 1A</xref>). PC2 separates the HOT promoters and HOT enhancers (<xref ref-type="fig" rid="fig2">Figure 2A</xref>, <xref ref-type="fig" rid="fig2s1">Figure 2—figure supplement 1B</xref>), and the PC1-PC2 combination also separates the p300-bound HOT loci (<xref ref-type="fig" rid="fig2">Figure 2B</xref>, <xref ref-type="fig" rid="fig2s1">Figure 2—figure supplement 1C</xref>). This indicates that the HOT promoters and HOT enhancers must have distinct signatures of DAPs. To test if such signatures exist, we clustered the DAPs according to the fractions of HOT promoter and HOT enhancer loci that they overlap with. This analysis showed that there is a large cluster of DAPs (n=458) which on average overlap with only 17% of HOT loci which are likely secondary to the HOT locus formation (<xref ref-type="fig" rid="fig2s2">Figure 2—figure supplement 2</xref>). We focused on the other, HOT-enriched, cluster of DAPs (n=87) which are present in 53% of HOT loci on average (<xref ref-type="fig" rid="fig2s2">Figure 2—figure supplement 2</xref>) and consist of four major clusters of DAPs (<xref ref-type="fig" rid="fig2">Figure 2D</xref>). <italic>Cluster I</italic> comprises four DAPs ZNF687, ARID4B, MAX, and SAP130 which are present in 75% of HOT loci on average. The three latter of these DAPs form a PPI interaction network (PPI enrichment p-value=0.001) (<xref ref-type="fig" rid="fig2s3">Figure 2—figure supplement 3A</xref>). We called this cluster of DAPs essential regulators given their widespread presence in both HOT enhancers and HOT promoters. <italic>Cluster II</italic> comprises 29 DAPs which are present in 47% of the HOT loci and are 1.7× more likely to overlap with HOT promoters than HOT enhancers. Among these DAPs are POLR2 subunits, PHF8, GABP1, GATAD1, TAF1, etc. The strongest associated GO molecular function term with the DAPs of this cluster is <italic>RNA Polymerase transcription factor initiation activity</italic> suggestive of their direct role in transcriptional activity (<xref ref-type="fig" rid="fig2s3">Figure 2—figure supplement 3B</xref>). <italic>Cluster III</italic> comprises 16 DAPs which are 1.9× more likely to be present in HOT enhancers than in HOT promoters. These are a wide variety of transcriptional regulators among which are those with high expression levels in liver NFIL3, NR2F6, and pioneer factors HNF4A, CEBPA, FOXA1, and FOXA2. The majority (13/16) of DAPs of this cluster form a PPI network (PPI enrichment p-value&lt;10<sup>–16</sup>, <xref ref-type="fig" rid="fig2s3">Figure 2—figure supplement 3C</xref>). Among the strongest associated GO terms of biological processes are those related to cell differentiation (<italic>white fat cell differentiation</italic>, <italic>endocrine pancreas development</italic>, <italic>dopaminergic neuron differentiation</italic>, etc.) suggesting that <italic>cluster III</italic> HOT enhancers underlie cellular development. <italic>Cluster IV</italic> comprises 12 DAPs which are equally abundant in both HOT enhancers and HOT promoters (64% and 63%, respectively), which form a PPI network (PPI enrichment p-value&lt;10<sup>–16</sup>, <xref ref-type="fig" rid="fig2s3">Figure 2—figure supplement 3D</xref>) with HDAC1 (histone deacetylase 1) being the node with the highest degree, suggesting that the DAPs of the cluster may be involved in chromatin-based transcriptional repression. Lastly, <italic>Cluster V</italic> comprises 26 DAPs of a wide range of transcriptional regulators, with a 1.3× skew toward the HOT enhancers. While this cluster contains prominent TFs such as TCF7L2, FOXA3, SOX6, FOSL2, etc., the variety of the pathways and interactions they partake in makes it difficult to ascertain the functional patterns from the constituent of DAPs alone. Although this clustering analysis reveals subsets of DAPs that are specific to either HOT enhancers or HOT promoters (Clusters II and III), it still does not explain what sorts of interplays take place between these recipes of HOT promoters and HOT enhancers, as well as with the other clusters of DAPs with equal abundance in both the HOT promoters and HOT enhancers.</p><fig-group><fig id="fig2" position="float"><label>Figure 2.</label><caption><title>PCA plots of high-occupancy target (HOT) loci based on the DNA-associated protein (DAP) presence vectors.</title><p>Each dot represents a HOT locus: (<bold>A</bold>) PC1 and PC2, marked promoters and enhancers. (<bold>B</bold>) PC1 and PC2, marked p300-bound HOT loci. (<bold>C</bold>) PC1 and PC4, marked CTCF-bound HOT loci. The dashed lines in A, B, C are logistic regression lines. auROC values are results of logistic regression. (<bold>D</bold>) DAPs hierarchically clustered by their involvement in HOT promoters and HOT enhancers. Heatmap colors indicate the % of HOT enhancers or promoters that a given DAP overlaps with. All of the visualized data is generated from the HepG2 cell line.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-95170-fig2-v1.tif"/></fig><fig id="fig2s1" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 1.</label><caption><title>PCA plots of high-occupancy target (HOT) loci in HepG2 based on the DNA-associated protein (DAP) presence vectors.</title><p>Each dot represents a HOT locus: (<bold>A</bold>) PC1 and PC2 correlated with the number of overlapping DAPs. (<bold>B</bold>) PC2 and PC3, with promoter and enhancer marked. (<bold>C</bold>) PC1 and PC2, marked p300-bound HOT loci. (<bold>D</bold>) PC1 and PC4, marked Cohesin-bound HOT loci.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-95170-fig2-figsupp1-v1.tif"/></fig><fig id="fig2s2" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 2.</label><caption><title>DNA-associated proteins (DAPs) clustered by percentage of high-occupancy target (HOT) promoters and HOT enhancers that the ChIP-seq peaks overlap.</title><p>The top cluster comprises the DAPs which on average overlap with 13% of HOT loci. The DAPs which form the bottom cluster are present in 53% of HOT loci.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-95170-fig2-figsupp2-v1.tif"/></fig><fig id="fig2s3" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 3.</label><caption><title>PPI networks of four clusters.</title><p>Names of the clusters are indicated as titles. Refer to the text for interpretations.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-95170-fig2-figsupp3-v1.tif"/></fig><fig id="fig2s4" position="float" specific-use="child-fig"><label>Figure 2—figure supplement 4.</label><caption><title>CTCF and Cohesin in high-occupancy target (HOT) loci.</title><p>(<bold>A</bold>) Distances to the nearest TSSs in HOT loci bound by CTCF and Cohesin. (<bold>B</bold>) Numbers of total DNA-associated proteins (DAPs) in HOT loci bound by CTCF and Cohesin.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-95170-fig2-figsupp4-v1.tif"/></fig></fig-group><p>Notably, PC4 separates HOT loci associated with CTCF (<xref ref-type="fig" rid="fig2">Figure 2C</xref>) and Cohesin (<xref ref-type="fig" rid="fig2s1">Figure 2—figure supplement 1D</xref>). This clear separation of CTCF- and Cohesin-bound HOTs is surprising, given that only relatively small fractions of their peaks (21% and 38%, respectively) reside in HOT loci, and present in 36% of the HOT loci, compared to some other DAPs with much higher presence described above, that do not get separated clearly by the PCA. Furthermore, CTCF- and Cohesin-bound HOT enhancer loci are located significantly closer (p-value&lt;10<sup>–100</sup>; Mann-Whitney U test) to the nearest genes (<xref ref-type="fig" rid="fig2s4">Figure 2—figure supplement 4A</xref>), making it more likely that those loci are proximal enhancers. And the total number of overlapping DAPs is significantly higher (p-value&lt;10<sup>–100</sup>; Mann-Whitney U test) in CTCF- and Cohesin-bound loci compared to the rest of the HOT loci (<xref ref-type="fig" rid="fig2s4">Figure 2—figure supplement 4B</xref>), suggesting that at least a portion of the number of DAPs in HOT loci can be explained by 3D chromatin contacts between the genomic regions mediated by CTCF-Cohesin complex.</p><p>To comprehensively quantify the 3D chromatin interactions involving the HOT loci, we used Hi-C data with 5 kb resolution (<xref ref-type="bibr" rid="bib32">Lieberman-Aiden et al., 2009</xref>) (see Methods). First, we obtained statistically significant chromatin interactions using FitHiChIP tool (<xref ref-type="bibr" rid="bib5">Bhattacharyya et al., 2019</xref>) (see Methods) and observed that HOT loci are enriched in chromatin interactions and 1.66× more likely to engage in chromatin interactions than the regular enhancers (p-value&lt;10<sup>–20</sup>, Chi-square test). When all of the DAP-bound loci are considered, the number of chromatin interactions positively correlates with the number of bound DAPs (rho = 0.3, p-value&lt;10<sup>–100</sup>, Spearman correlation). Next, we overlayed the chromatin interactions with the loci binned by the number of bound DAPs. We observed that the loci with high numbers of bound DAPs are more likely to engage in chromatin interactions with other loci harboring large numbers of DAPs, i.e., the HOT loci have the propensity to connect through long-range chromatin interactions with other HOT loci (<xref ref-type="fig" rid="fig3">Figure 3A</xref>). To further validate this observation, we obtained frequently interacting regions (FIREs) (<xref ref-type="bibr" rid="bib53">Schmitt et al., 2016</xref>), and observed that the FIREs are 2.89× (p-value&lt;10<sup>–230</sup>, Chi-square test) enriched HOT loci compared to the regular enhancers (see Methods). Moreover, 66% of HOT loci are located in TAD regions and 21% are located in chromatin loops. In particular, the HOT loci are 2.97× (p-value&lt;10<sup>–230</sup>, Mann-Whitney U test) enriched in the chromatin loop anchor regions (11% of the HOT loci) compared to regular enhancers. To investigate further, we analyzed the loop anchor regions harboring HOT loci and observed that the number of multi-way contacts on loop anchors (i.e. loci that serve as anchors to multiple loops) correlates with the number of bound DAPs (rho = 0.84 p-value&lt;10<sup>–4</sup>; Pearson correlation). The number of multi-way interactions in loop anchor regions varies between 1 and 6, with only one locus, in an extreme case, serving as an anchor for 6 overlapping loops on chromosome 2 (<xref ref-type="fig" rid="fig3">Figure 3B</xref>). Of the loop anchor regions with &gt;3 overlapping loops, more than half contained at least one HOT locus, suggesting an interplay between chromatin loops and HOT loci (<xref ref-type="fig" rid="fig3">Figure 3B</xref>). Overall, 94% of HOT loci are located in regions with at least one chromatin interaction. This observation is consistent with previous reports that much of the long-range 3D chromatin contacts form through the interactions of large protein complexes (<xref ref-type="bibr" rid="bib47">Quinodoz et al., 2018</xref>). While there is a correlation between the HOT loci and chromatin interactions, the causal relation between these two properties of genomic loci is not clear.</p><fig id="fig3" position="float"><label>Figure 3.</label><caption><title>High-occupancy target (HOT) loci in high-frequency 3D chromatin interaction regions.</title><p>(<bold>A</bold>) Densities of long-range Hi-C chromatin contacts between the DNA-associated protein (DAP)-bound loci. Each horizontal and vertical bin represents the loci with the number of bound DAPs between the edge values. The density values of each cell are normalized by the maximum value across all pairwise bins. Green boxes represent HOT loci. (<bold>B</bold>) Distribution of HOT loci in Hi-C contact regions. X-axis is the number of Hi-C contacts. Numbers in the top row indicate the total number of genomic loci engaging in the given number of Hi-C contacts. Bars indicate the % of Hi-C loci that contain at least one HOT locus. (<bold>C</bold>) Distribution of the number of HOT loci in regions with a given number of Hi-C contacts. X-axis is the same as B. All of the visualized data is generated from the HepG2 cell line.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-95170-fig3-v1.tif"/></fig></sec><sec id="s2-2"><title>A set of DAPs stabilizes the interactions of DAPs at HOT loci</title><p>Next, we sought to analyze the patterns of ChIP-seq signal values at HOT loci, as a metric for overall DAP occupancy at genomic loci. We observed that the overall signals of DAPs correlate with the total number of colocalizing DAPs (<xref ref-type="fig" rid="fig4">Figure 4A</xref>, rho = 0.97, p-value&lt;10<sup>–10</sup>; Spearman correlation). Moreover, even when calculated DAP-wise, the average of the overall signal strength of every DAP correlates with the fraction of HOT loci that the given DAP overlaps with (rho = 0.6, p-value&lt;10<sup>–29</sup>; Spearman correlation, <xref ref-type="fig" rid="fig4">Figure 4B</xref>), meaning that the overall average value of the signal intensity of a given DAP is largely driven by the ChIP-seq peaks which are located in HOT loci.</p><fig-group><fig id="fig4" position="float"><label>Figure 4.</label><caption><title>High-occupancy target (HOT) regions induce strong ChIP-seq signals.</title><p>(<bold>A</bold>) Distribution of the signal values of the ChIP-seq peaks by the number of bound DNA-associated proteins (DAPs). The shaded region represents the HOT loci. (<bold>B, C</bold>) DAPs sorted by the ratio of ChIP-seq signal strength of the peaks located in HOT loci and non-HOT loci. 20 most HOT-specific (red bars) and 20 most non-HOT-specific (blue bars) DAPs are depicted. (<bold>B</bold>) Fold-change (log2) of the HOT and non-HOT loci ChIP-seq signals. (<bold>C</bold>) Distribution of the average ChIP-seq signal in the loci binned by the number of bound DAPs. Rows represent the loci with the bound DAPs indicated by the values of the edges (y-axis). Green box regions demarcate the HOT regions. (<bold>D</bold>) Signal values of sequence-specific DAPs (ssDAPs), non-sequence-specific DAPs (nssDAPs) (see the text for description), H3K27ac, CTCF, P300 peaks in HOT promoters and enhancers. All of the visualized data is generated from the HepG2 cell line.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-95170-fig4-v1.tif"/></fig><fig id="fig4s1" position="float" specific-use="child-fig"><label>Figure 4—figure supplement 1.</label><caption><title>Normalized ChIP-seq signal values of DNA-associated proteins (DAPs) in high-occupancy target (HOT) loci (rows) in the presence of other DAPs (columns).</title><p>The hierarchical clustering is done using the columns. That is, the leftmost outer group (in the green box) contains the DAPs in the presence of which most of the other DAPs yield highest ChIP-seq signal values.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-95170-fig4-figsupp1-v1.tif"/></fig><fig id="fig4s2" position="float" specific-use="child-fig"><label>Figure 4—figure supplement 2.</label><caption><title>Distribution of the ChIP-seq signal values of DNA-associated proteins (DAPs) when the stabilizing DAPs (<xref ref-type="fig" rid="fig4s1">Figure 4—figure supplement 1</xref>) are present vs. absent.</title></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-95170-fig4-figsupp2-v1.tif"/></fig></fig-group><p>While the overall average of the ChIP-seq signal intensity in HOT loci is greater when compared to the rest of the DAP-bound loci, individual DAPs demonstrate different levels of involvement in HOT loci. When sorted by the ratio of the signal intensities in HOT vs. non-HOT loci, among those with the highest HOT-affinities are GATAD1, MAX, NONO, as well as POLR2G and Mediator subunit MED1 (<xref ref-type="fig" rid="fig4">Figure 4B and C</xref>). Whereas those with the opposite affinity (i.e. those that have the strongest binding sites in non-HOT loci) are REST, RFX5, TP53, etc. (<xref ref-type="fig" rid="fig4">Figure 4B and C</xref>). By analyzing the signal strengths of DAPs jointly, we observed that a host of DAPs likely has a stabilizing effect on the binding of DAPs in that, when present, the signal strengths of the majority of DAPs are on average 1.9× greater (p-value&lt;10<sup>–100</sup>, Mann-Whitney U test). These DAPs are CREB1, RFX1, ZNF687, RAD51, ZBTB40, and GPBP1L1 (Appendix 1 – Joint DAPs analysis, <xref ref-type="fig" rid="fig4s1">Figure 4—figure supplements 1</xref> and <xref ref-type="fig" rid="fig4s2">2</xref>).</p><p>So far, we have treated the DAPs under a single category and did not make a distinction based on their known DNA-binding properties. Previous studies have discussed the idea that sequence-specific DAPs (ssDAPs) can serve as anchors, similar to the pioneer TFs, which could facilitate the formation of HOT loci (<xref ref-type="bibr" rid="bib48">Ramaker et al., 2020</xref>; <xref ref-type="bibr" rid="bib45">Partridge et al., 2020</xref>; <xref ref-type="bibr" rid="bib73">Xie et al., 2013</xref>). We asked if ssDAPs yield greater signal strength values than non-sequence-specific DAPs (nssDAPs). To test this hypothesis, we classified the DAPs into those two categories using the definitions provided in the study (<xref ref-type="bibr" rid="bib27">Lambert et al., 2018</xref>), where the TFs are classified by curation through extensive literature review and supported by annotations such as the presence of DNA-binding domains and validated binding motifs. Based on this classification, we categorized the ChIP-seq signal values into these two groups. While statistically significant (p-value&lt;0.001, Mann-Whitney U test), the differences in the average signals of ssDAPs and nssDAPs in both HOT enhancers and HOT promoters are small (<xref ref-type="fig" rid="fig4">Figure 4D</xref>). Moreover, while the average signal values of ssDAPs in HOT enhancers are greater than that of the nssDAPs, in HOT promoters this relation is reversed. At the same time, the average signal strength of the DAPs is 3× greater than the average signal strength of H3K27ac peaks in HOT loci. Based on this, we concluded that the ChIP-seq signal intensities do not seem to be a function of the DNA-binding properties of the DAPs.</p></sec><sec id="s2-3"><title>Sequence features that drive the accumulation of DAPs</title><p>We next analyzed the sequence features of the HOT loci. For this purpose, we first addressed the evolutionary conservation of the HOT loci using phastCons scores generated using an alignment of 46 vertebrate species (<xref ref-type="bibr" rid="bib57">Siepel et al., 2005</xref>). The average conservation scores of the DAP-bound loci are in strong correlation with the number of bound DAPs (rho = 0.98, p-value&lt;10<sup>–130</sup>; Spearman correlation), indicating that the negative selection exerted on HOT loci are proportional to the number of bound DAPs (<xref ref-type="fig" rid="fig5">Figure 5A</xref>). With 120 DAPs per locus on average, these HOT regions are 1.7× more conserved than the regular enhancers in HepG2 (<xref ref-type="fig" rid="fig5">Figure 5B</xref>). We observed a similar trend of conservation levels when the phastCons scores generated from primates and placental mammals and primates were considered, the HOT loci being 1.45× and 1.1× more conserved than the regular enhancers, respectively (<xref ref-type="fig" rid="fig5s1">Figure 5—figure supplement 1</xref>). In addition, we observed that the HOT loci of all three cell lines (HepG2, K562, and H1) overlap with 22 ultraconserved regions, among which are the promoter regions of 11 genes including SP5, SOX5, AUTS2, PBX1, ZFPM2, ARID1A, OLA1 and the enhancer regions of (within &lt;50 kb of their TSS) 5S rRNA, MIR563, SOX21, etc. (full list in <xref ref-type="supplementary-material" rid="supp1">Supplementary file 1, table S4</xref>). Among them are those which have been linked to diseases and other phenotypes. For example, DNAJC1 (<xref ref-type="bibr" rid="bib37">Michailidou et al., 2017</xref>) and OLA1 (which interacts with BRCA1) have been linked to breast cancer in cancer GWAS studies (<xref ref-type="bibr" rid="bib33">Liu et al., 2020</xref>). Whereas AUTS2 (<xref ref-type="bibr" rid="bib6">Biel et al., 2022</xref>) and SOX5 (<xref ref-type="bibr" rid="bib52">Schanze et al., 2013</xref>) have been linked to predisposition to neurological conditions such as autism spectrum disorder, intellectual disability, and neurodevelopmental disorder. Of these genes, ARID1A, AUTS2, DNAJC1, OLA1, SOX5, and ZFPM2 have been reported to have strong activities in the Allen Mouse Brain Atlas (<xref ref-type="bibr" rid="bib12">Daigle et al., 2018</xref>).</p><fig-group><fig id="fig5" position="float"><label>Figure 5.</label><caption><title>Sequence features of high-occupancy target (HOT) loci.</title><p>(<bold>A</bold>) Distribution of conservation score in loci bound by DNA-associated proteins (DAPs) in HepG2 and K562. The logarithmic part of the bins is expressed in terms of the percentages of loci that each bin covers, averaged over two cell lines. The shaded region represents HOT loci. (<bold>B</bold>) phastCons conservation scores of regular enhancer, HOT loci, and exon regions. The values are normalized by the average scores of regular enhancers. (<bold>C</bold>) Classification performances (auROC) of HOT loci against the backgrounds of DNase-I hypersensitivity sites (DHS), promoter, and regular enhancer regions. The x-axis values are the methods used for classifications. Methods starting with ‘seq -’ are based on sequences (convolutional neural networks [CNNs] and gkmSVM). Starting with ‘feat -’ are methods where all sequence features are used (GC, CpG, GpC, CpG island).</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-95170-fig5-v1.tif"/></fig><fig id="fig5s1" position="float" specific-use="child-fig"><label>Figure 5—figure supplement 1.</label><caption><title>Comparison of phastCons conservation scores of regular enhancers, high-occupancy target (HOT) loci, and exons using the score extracted from vertebrates, placental mammals, and primates.</title></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-95170-fig5-figsupp1-v1.tif"/></fig><fig id="fig5s2" position="float" specific-use="child-fig"><label>Figure 5—figure supplement 2.</label><caption><title>Sequence features of high-occupancy target (HOT) loci.</title><p>(<bold>A</bold>) Fractions of DNA-associated protein (DAP)-bound loci overlapping CpG islands. X-axis is bins of number of bound DAPs. The logarithmic bins are represented in terms of percent of total number of DAPs in given cell line. (<bold>B</bold>) GC contents of DAP-bound loci. X-axis is the same as in A. (<bold>C</bold>) Fractions of loci DAP-bound overlapping repeat elements. X-axis is the same as in A.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-95170-fig5-figsupp2-v1.tif"/></fig><fig id="fig5s3" position="float" specific-use="child-fig"><label>Figure 5—figure supplement 3.</label><caption><title>Comparison of classification performances for sequences in different lengths.</title></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-95170-fig5-figsupp3-v1.tif"/></fig><fig id="fig5s4" position="float" specific-use="child-fig"><label>Figure 5—figure supplement 4.</label><caption><title>Distances and expressions of flanking genes of DNA-associated protein (DAP)-bound loci.</title><p>(<bold>A</bold>) Expression levels of target genes of DAP-bound loci in HepG2. (<bold>B</bold>) Distance to the nearest TSS from the DAP-bound non-promoter loci in HepG2 and K562.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-95170-fig5-figsupp4-v1.tif"/></fig></fig-group><p>CpG islands have been postulated to serve as permissive TF binding platforms (<xref ref-type="bibr" rid="bib42">Pachano et al., 2021</xref>; <xref ref-type="bibr" rid="bib14">Deaton and Bird, 2011</xref>) and this has been listed as one of the possible reasons for the existence of HOT loci in a previous study (<xref ref-type="bibr" rid="bib71">Wreczycka et al., 2019</xref>). To test this hypothesis, we extracted the overlap rates of all DAP-bound loci with CpG islands (Methods). While the overall fraction of loci that overlap CpG islands correlates strongly with the number of bound DAPs (rho = 0.7, p-value=0.001; Pearson correlation), only 12% of HOT enhancers overlapped CpG island whereas, for the HOT promoters, this fraction was 83%, suggesting that CpG islands alone do not explain HOT enhancer loci despite accounting for the majority of HOT promoters loci (<xref ref-type="fig" rid="fig5s2">Figure 5—figure supplement 2A</xref>). Similarly, the average GC content is strongly correlated with the number of bound DAPs (rho = 0.89, p-value&lt;10<sup>–4</sup>; Pearson correlation, <xref ref-type="fig" rid="fig5s2">Figure 5—figure supplement 2B</xref>), with the average GC content of 64% and 51% in HOT promoters and HOT enhancers respectively (p-value&lt;10<sup>–100</sup>, Mann-Whitney U test), in both HepG2 and K562.</p><p>In addition, we observed that the average content of repeat elements in the loci strongly and negatively correlates with the number of bound DAPs across the cell lines (rho = −0.9, p-value=&lt;10<sup>–5</sup>; Pearson, <xref ref-type="fig" rid="fig5s2">Figure 5—figure supplement 2C</xref>), which is likely the result of the fact that the HOTs are under elevated negative selection and reject insertion of repetitive DNA.</p><p>Other genomic sequence features that have been considered in the context of HOT loci in previous studies include and are not limited to G-quadruplex, R-loops, methylation patterns, etc., which have concluded that each of them can partially explain the phenomenon of the HOT loci (<xref ref-type="bibr" rid="bib40">Moorman et al., 2006</xref>; <xref ref-type="bibr" rid="bib60">Teytelman et al., 2013</xref>; <xref ref-type="bibr" rid="bib71">Wreczycka et al., 2019</xref>). Still, one of the central questions remains whether the HOT loci are driven by sequence features or they are the result of cellular biology not strictly related to the sequences, such as the proximal accumulation of DAPs in foci due to the biochemical properties of accumulated molecules, or other epigenetic mechanisms.</p><p>To address this question with a broader approach, we asked whether the HOT loci can be accurately predicted based on their DNA sequences alone, and sequence features, including GC, CpG, GpC contents, and CpG island coverage. For sequence-based classification, we trained a convolutional neural network (CNN) model using one-hot encoded sequences and an SVM classifier trained on gapped k-mers (seq-SVM) (<xref ref-type="bibr" rid="bib28">Lee, 2016</xref>). Using the sequence features we trained SVM models with linear kernel function (feature-SVM). We carried out the classification experiments using the following control (i.e. negative) sets: (a) randomly selected loci from merged DNase I hypersensitivity sites (DHS) of cell lines in the Roadmap Epigenomics Project, (b) promoter regions, and (c) regular enhancers. When averaged over cell lines and control sets, CNN, seq-SVM, and feature-SVM models yielded auROC values of 0.91, 0.86, and 0.78 respectively, suggesting that CNNs capture the motif grammar of the HOT loci better than the compared models (<xref ref-type="fig" rid="fig5">Figure 5C</xref>). The superiority of sequence-based models over feature-based classification by a factor of 1.3× (or 17%) suggests that there is additional information that is highly relevant to the DNA-DAP interaction density encoded in the DNA sequences, in addition to the GC, CpG, GpC contents. (See Appendix 1 – Classification results analyses for further details of model training, and comparison of performances of different combinations of SVM kernels and feature sets, as well as Logistic Regression as a baseline.) This is in line with the observation mentioned above, that 88% of the HOT enhancers do not overlap with annotated CpG islands. This analysis concluded that the mechanisms of HOT locus formation are likely encoded in their DNA sequences.</p><p>Extending the input regions from 400 bp to 1 kb for sequence-based classification did not lead to a significant increase in performance, suggesting that the core 400 bp regions contain most of the information associated with DAP density (<xref ref-type="fig" rid="fig5s3">Figure 5—figure supplement 3</xref>).</p></sec><sec id="s2-4"><title>Highly expressed housekeeping genes are commonly regulated by HOT promoters</title><p>After characterizing the HOT loci in terms of the DAP composition and sequence features, we sought to analyze the cellular processes they partake in. HOT loci were previously linked to highly expressed genes (<xref ref-type="bibr" rid="bib71">Wreczycka et al., 2019</xref>). In both inspected differentiated cell lines (HepG2 and K562), the number of DAPs positively correlates with the expression level of their target gene (enhancers were assigned to their nearest genes for this analysis; rho = 0.56, p-value&lt;10<sup>–10</sup>; Spearman correlation; <xref ref-type="fig" rid="fig5s4">Figure 5—figure supplement 4A</xref>). In HepG2, the average expression level of the target genes of promoters with at least one DAP bound is 1.7× higher than that of the target genes of enhancers with at least one DAP bound, whereas when only HOT loci are considered this fold-increase becomes 4.7×. This suggests that the number of bound DAPs of the HOT locus has a direct impact on the level of the target gene expression. Moreover, highly expressed genes (RPKM&gt;50) were 4× more likely to have multiple HOT loci within the 50 kb of their TSSs than the genes with RPKM&lt;5 (p-value&lt;10<sup>–12</sup>, Chi-square test). In addition, the average distance between HOT enhancer loci and the nearest gene is 4.5× smaller than with the regular enhancers (p-value&lt;10<sup>–30</sup>, Mann-Whitney U test). Generally, we observed that the distances between the HOT enhancers and the nearest genes are negatively correlated with the number of bound DAPs (rho = −0.9; p-value&lt;10<sup>–6</sup>; Pearson correlation; <xref ref-type="fig" rid="fig5s4">Figure 5—figure supplement 4B</xref>), suggesting that the increasing number of bound DAPs makes the regulatory region more likely to be the TSS-proximal regulatory region.</p><p>To further analyze the distinction in involved biological functions between the HOT promoters and enhancers, we compared the fraction of housekeeping (HK) genes that they regulate, using the list of HK genes reported by <xref ref-type="bibr" rid="bib22">Hounkpe et al., 2021</xref>. According to this definition, 64% of HK genes are regulated by a HOT promoter and only 30% are regulated by regular promoters (<xref ref-type="fig" rid="fig6">Figure 6A</xref>). The HOT enhancers, on the other hand, flank 21% of the HK genes, which is less than the percentage of HK genes flanked by regular enhancers (38%). For comparison, 22% of the flanking genes of super-enhancers constitute HK genes. The involvement of HOT promoters in the regulation of HK genes is also confirmed in terms of the fraction of loci flanking the HK genes, namely, 21% of the HOT promoters regulate 64% of the HK genes. This fraction is much smaller (&lt;9% on average) for the rest of the mentioned categories of loci (HOT and regular enhancers, regular promoters, and super-enhancers, <xref ref-type="fig" rid="fig6">Figure 6A</xref>).</p><fig id="fig6" position="float"><label>Figure 6.</label><caption><title>High-occupancy target (HOT) promoters are ubiquitous and HOT enhancers are tissue-specific.</title><p>(<bold>A</bold>) Fractions of housekeeping genes regulated by the given category of loci (blue). Fractions of the loci which regulate the housekeeping genes (orange). (<bold>B</bold>) Tissue specificity (<italic>tau</italic>) scores of the target genes of different types of regulatory regions. (<bold>C</bold>) GO enriched terms of HOT promoters and enhancers of HepG2. 0 values in the p-values columns indicate that the GO term was not present in the top 50 enriched terms as reported by the GREAT tool. All of the visualized data is generated from the HepG2 cell line.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-95170-fig6-v1.tif"/></fig><p>We then asked whether the tissue specificities of the expression levels of target genes of the HOT loci reflect their involvement in the regulation of HK genes. For this purpose, we used the <italic>tau</italic> metric as reported by <xref ref-type="bibr" rid="bib44">Palmer et al., 2021</xref>, where a high <italic>tau</italic> score (between 0 and 1) indicates a tissue-specific expression of a gene, whereas a low <italic>tau</italic> score means that the transcript is expressed stably across tissues. We observed that the average <italic>tau</italic> scores of target genes of HOT enhancers are significantly but by a small margin greater than the regular enhancers (0.66 and 0.63, respectively; p-value&lt;10<sup>–18</sup>, Mann-Whitney U test), with super-enhancers being equal to regular enhancers (0.63). The difference in the average <italic>tau</italic> scores of the HOT and regular promoters is stark (0.57 and 0.74, respectively, p-value&lt;10<sup>–100</sup>, Mann-Whitney U test), representing a 23% increase (<xref ref-type="fig" rid="fig6">Figure 6B</xref>). Combined with the involvement in the regulation of HK genes, average <italic>tau</italic> scores suggest that the HOT promoters are more ubiquitous than the regular promoters whereas HOT enhancers are more tissue-specific than the regular and super-enhancers. Further supporting this, the GO enrichment analysis showed that the GO terms associated with the set of genes regulated by HOT promoters are basic HK cellular functions (such as <italic>RNA processing</italic>, <italic>RNA metabolism</italic>, <italic>ribosome biogenesis,</italic> etc.), whereas HOT enhancers are enriched in GO terms of cellular response to the environment and liver-specific processes (such as <italic>response to insulin, oxidative stress, epidermal growth factors,</italic> etc.) (<xref ref-type="fig" rid="fig6">Figure 6C</xref>).</p></sec><sec id="s2-5"><title>A core set of HOT loci is active during development which expands after differentiation</title><p>Having observed that the HOT loci are active regions in many other human cell types, we asked if the observations made on the HOT loci of differentiated cell lines also hold true in the embryonic stage. To that end, we analyzed the HOT loci in H1 cells. It is important to note that the number of available DAPs in H1 cells is significantly smaller (n=47) than in HepG2 and K562, due to a much smaller size of the ChIP-seq dataset generated in H1. Therefore, the criterion of having &gt;17% of available DAPs yields n&gt;15 DAPs for the H1, as opposed to 77 and 55 for HepG2 and K562, respectively. However, many of the features of the loci that we’ve analyzed so far demonstrated similar patterns (GC contents, target gene expressions, ChIP-seq signal values, etc.) when compared to the DAP-bound loci in HepG2 and K562, suggesting that albeit limited, the distribution of the DAPs in H1 likely reflects the true distribution of HOT loci. To alleviate the difference in available DAPs, in addition to comparing the HOT loci defined using the complete set of DAPs, we also (a) applied the HOT classification routing using a set of DAPs (n=30) available in all three cell lines, (b) randomly subselected DAPs in HepG2 and K562 to match the number of DAPs in H1.</p><p>We observed that, when the complete set of DAPs is used, 85% of the HOT loci of H1 are also HOT loci in either of the other two differentiated cell lines (<xref ref-type="fig" rid="fig7">Figure 7A</xref>). However, only &lt;10% of the HOT loci of the two differentiated cell lines overlapped with H1 HOT loci, suggesting that the majority of the HOT loci are acquired after the differentiation. A similar overlap ratio was observed based on DAPs common to all three cell lines (<xref ref-type="fig" rid="fig7">Figure 7B</xref>), where 68% of H1 HOT loci overlapped with that of the differentiated cell lines. These overlap levels were much higher than the randomly selected DAPs matching the H1 set (30%, <xref ref-type="fig" rid="fig7">Figure 7C</xref>).</p><fig-group><fig id="fig7" position="float"><label>Figure 7.</label><caption><title>H1-hESC high-occupancy target (HOT) loci.</title><p>(<bold>A</bold>) Overlaps between the HOT loci of three cell lines. (<bold>B</bold>) Overlaps between the HOT loci of cell lines defined using the set of DNA-associated proteins (DAPs) available in all three cell lines. (<bold>C</bold>) Fractions of H1 HOT loci overlapping with that of the HepG2 and K562 using the complete set of DAPs, common DAPs, and DAPs randomly subsampled in HepG2/K562 to match the size of H1 DAPs set. (<bold>D</bold>) phastCons scores of HOT loci in HepG2, K562, and H1.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-95170-fig7-v1.tif"/></fig><fig id="fig7s1" position="float" specific-use="child-fig"><label>Figure 7—figure supplement 1.</label><caption><title>GO terms associated with the high-occupancy target (HOT) enhancers and promoters in H1-hESC.</title></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-95170-fig7-figsupp1-v1.tif"/></fig></fig-group><p>Average evolutionary conservation scores (phastCons) of the developmental HOT loci are 1.3× higher than K562 and HepG2 HOT loci (p-value&lt;10<sup>–10</sup>, Mann-Whitney U test, <xref ref-type="fig" rid="fig7">Figure 7D</xref>). It is conceivable to hypothesize that the embryonic HOT loci are located mainly in regions with higher conservation regions, and more regulatory regions emerge as HOT loci after the differentiation. Some of these tissue-specific HOT loci could be those that are acquired more recently (compared to the H1 HOT loci), as it is known that the enhancers are often subject to higher rates of evolutionary turnover than the promoters (<xref ref-type="bibr" rid="bib15">Domené et al., 2013</xref>).</p><p>GO enrichment analysis showed that H1 HOT promoters, similarly to the other cell lines, regulate the basic HK processes (<xref ref-type="fig" rid="fig7s1">Figure 7—figure supplement 1</xref>) while the HOT enhancers regulate responses to environmental stimuli and processes active during the embryonic stage such as <italic>TORC1 signaling</italic> and <italic>beta-catenin-TCF assembly</italic>. This suggests that the main processes that the HOT promoters are involved in during the development remain relatively unchanged after the differentiation (in terms of associated GO terms, and due to being the same loci as the HOT promoters in differentiated cell lines), whereas the scope of the cellular activities regulated by HOT enhancers gets expanded after differentiation to be more exclusively tissue-specific.</p></sec><sec id="s2-6"><title>HOT loci are enriched in causal variants</title><p>After establishing the expression and tissue specificities of the HOT loci, we next analyzed the polymorphic variability in HOT loci and whether these loci are enriched in phenotypically causal variants. First, we analyzed the density of common variants extracted from the gnomAD database (<xref ref-type="bibr" rid="bib25">Karczewski et al., 2020</xref>) (filtered with MAF&gt;5%). We observed that HOT enhancers and HOT promoters are depleted in INDELs (4.7 and 4.1 variants per 1 kb, respectively), compared to the regular enhancers and regular promoters (5.5 and 6.2 variants per 1 kb, p-value&lt;10<sup>–4</sup> and &lt;10<sup>–100</sup>, respectively, Mann-Whitney U test; <xref ref-type="fig" rid="fig8">Figure 8A</xref>). Contradicting the pattern of conservation scores described above, the distribution of common SNPs is elevated in HOT enhancers and HOT promoters compared to regular enhancers and regular promoters (1.14× and 1.07× fold-enrichment, p-values&lt;10<sup>–20</sup> and &lt;10<sup>–100</sup>, respectively, Mann-Whitney U test; <xref ref-type="fig" rid="fig8">Figure 8B</xref>). This elevation of common variants in HOT loci, despite being located in conserved loci, has been reported in a previous study in which the binding motifs of TFs were observed to colocalize in regions where the density of common variants was higher than average (<xref ref-type="bibr" rid="bib63">Vierstra et al., 2020</xref>).</p><fig-group><fig id="fig8" position="float"><label>Figure 8.</label><caption><title>Densities of variants.</title><p>(<bold>A</bold>) Common INDELs (MAF&gt;5%), (<bold>B</bold>) common SNPs (MAF &gt;5%), (<bold>C</bold>) eQTLs, (<bold>D</bold>) chromatin accessibility QTLs (caQTLs), (<bold>E</bold>) reporter array QTLs (raQTLs), and (<bold>F</bold>) GWAS and LD (r2&gt;0.8) variants in high-occupancy target (HOT) loci and regular promoters and enhancers. (<bold>G</bold>) Enriched GWAS traits in HOT enhancers and promoters. All of the visualized data is generated from the HepG2 cell line.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-95170-fig8-v1.tif"/></fig><fig id="fig8s1" position="float" specific-use="child-fig"><label>Figure 8—figure supplement 1.</label><caption><title>GWAS traits enrichment analysis filtered by unadjusted p-values (p-value&lt;0.001).</title></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-95170-fig8-figsupp1-v1.tif"/></fig></fig-group><p>The eQTLs, on the other hand, are 2.0× enriched in HOT promoters compared to the regular promoters (p-value&lt;10<sup>–21</sup>, Mann-Whitney U test), while HOT enhancers are only moderately enriched in eQTLs compared to the regular enhancers (1.15×, p-value&gt;0.05, Mann-Whitney U test; <xref ref-type="fig" rid="fig8">Figure 8C</xref>). eQTL enrichment in HOT promoters and regular promoters (compared to HOT and regular enhancers, respectively) is in line with the known characteristics of the eQTL dataset, that the eQTLs most commonly reflect TSS-proximal gene-variant relationships, and therefore are enriched in promoter regions since the TSS-distal eQTLs are hard to detect due to the burden of multiple tests (<xref ref-type="bibr" rid="bib10">Consortium, 2015</xref>).</p><p>Unlike the eQTL analysis, we observed that the chromatin accessibility QTLs (caQTLs) are dramatically enriched in the overall enhancer regions (HOT and regular) compared to the promoters (HOT and regular) (4.1×, p-value&lt;10<sup>–100</sup>; Mann-Whitney U test, <xref ref-type="fig" rid="fig8">Figure 8D</xref>). This observation confirms the findings of the study which reported the caQTL dataset in HepG2 cells (<xref ref-type="bibr" rid="bib11">Currin et al., 2021</xref>), which reported that the likely causal caQTLs are predominantly the variants disrupting the binding motifs of liver-expressed TFs enriched in liver enhancers. However, within the promoters regions, the HOT promoters are 3.0× enriched in caQTLs compared to the regular promoters (p-value=0.001; Mann-Whitney U test), whereas the fold enrichment in HOT enhancers is insignificant (1.2×, p-value=0.22, Mann-Whitney U test).</p><p>A similar enrichment pattern displays the reporter array QTLs (raQTLs; <xref ref-type="bibr" rid="bib62">van Arensbergen et al., 2019</xref>), with respect to the overall (HOT and regular) promoter and enhancer regions, with 3.3× enrichment in enhancers (p-value&lt;10<sup>–10</sup>, Mann-Whitney U test, <xref ref-type="fig" rid="fig8">Figure 8E</xref>). But, within-promoters and within-enhancers enrichments show that the enrichment in HOT promoters is more pronounced than the HOT enhancers (3.6× and 1.8×, p-values&lt;0.01 and&lt;10<sup>–11</sup>, respectively, Mann-Whitney U test). The enrichment of the raQTLs in enhancers over the promoters likely reflects the fact that the SNP-containing loci are first filtered for raQTL detection according to their capacities to function as enhancers in the reporter array (<xref ref-type="bibr" rid="bib62">van Arensbergen et al., 2019</xref>).</p><p>Combined, all three QTL datasets show a pronounced enrichment in HOT promoters compared to the regular promoters, whereas only the raQTLs show significant enrichment in HOT enhancers. This suggests that the individual DAP ChIP-seq peaks in HOT promoters are more likely to have consequential effects on promoter activity if altered, while HOT enhancers are less susceptible to mutations. Additionally, it is noteworthy that only the raQTLs are the causal variants, whereas e/caQTLs are correlative quantities subject to the effects of LD.</p><p>Finally, we used the GWAS SNPs combined with the LD SNPs (r2&gt;0.8) and observed that the HOT promoters are significantly enriched in GWAS variants (1.8×, p-value&gt;10<sup>–100</sup>) whereas the HOT enhancers show no significant enrichment over regular enhancers (p-value&gt;0.1, Mann-Whitney U test) (<xref ref-type="fig" rid="fig8">Figure 8F</xref>). We then calculated the fold-enrichment levels of GWAS traits SNPs using the combined DHS regions of Roadmap Epigenome cell lines as a background (see Methods). Filtering the traits with significant enrichment in HOT loci (p-value&lt;0.001, Binomial test, Bonferroni corrected, see Methods) left seven traits, of which all are definitively related to the liver functions (<xref ref-type="fig" rid="fig8">Figure 8G</xref>). Of the seven traits, only one (<italic>Blood protein level</italic>) was significantly enriched in regular promoters. While the regular enhancers are enriched in most of the (six of seven) traits, the overall enrichment values in HOT enhancers are 1.3× greater compared to the regular enhancers. The fold-increase is even greater (1.5×) between the HOT and DHS regions. When the enrichment significance levels are selected using unadjusted p-values, we obtained 24 GWAS traits, of which 22 are related to liver functions (<xref ref-type="fig" rid="fig8s1">Figure 8—figure supplement 1</xref>). This analysis demonstrated that the HOT loci are important for phenotypic homeostasis.</p></sec><sec id="s2-7"><title>Transcriptional condensates as a model for explaining the HOT regions</title><p>Recent studies on phase-separated condensates have established that condensates are ubiquitous in cells and play crucial roles in gene regulation through transcriptional condensates (<xref ref-type="bibr" rid="bib41">Nair et al., 2019</xref>; <xref ref-type="bibr" rid="bib29">Lee et al., 2022</xref>; <xref ref-type="bibr" rid="bib16">Feric and Misteli, 2022</xref>; <xref ref-type="bibr" rid="bib2">Ahn et al., 2021</xref>). We postulated that the HOT loci could be explainable if it can be shown that the HOT loci demonstrate a high propensity for the formation of transcriptional condensates. The hallmarks of transcriptional condensates include (not limited to) scaffolding proteins that undergo liquid-to-liquid phase separation (LLPS), DNA and RNA molecules, and intrinsically disordered (IDR) proteins. We sought to analyze whether these properties can be attributed to the HOT loci.</p><p>First, using CD-CODE database (<xref ref-type="bibr" rid="bib50">Rostam et al., 2023</xref>) we annotated 24% of the DAPs used in the analysis as LLPS-inducing proteins (<xref ref-type="fig" rid="fig9">Figure 9A</xref>). We observed that LLPS proteins are uniformly distributed in HOT loci (<xref ref-type="fig" rid="fig9">Figure 9B</xref>). We calculated a null distribution by randomly shuffling the ChIP-seq peaks in HOT loci 10 times, which resulted in a near-zero fraction of LLPS proteins located in &gt;45% of the HOT loci, where the actual observed fraction is 23% (average of the last two bins in <xref ref-type="fig" rid="fig9">Figure 9B</xref>), strongly suggesting an overrepresentation. Moreover, LLPS proteins yield significantly stronger ChIP-seq signals compared to the rest of the DAPs (<xref ref-type="fig" rid="fig9">Figure 9C</xref>, p-value=0.002, t-test), and contain a higher percentage of predicted IDR regions (<xref ref-type="fig" rid="fig9">Figure 9D</xref>, 30% vs. 26%, p-value=0.01, t-test).</p><fig id="fig9" position="float"><label>Figure 9.</label><caption><title>High-occupancy target (HOT) loci as transcriptional condensates.</title><p>(<bold>A</bold>) Fraction of DNA-associated proteins (DAPs) annotated as liquid-to-liquid phase separation (LLPS) proteins in CD-CODE database. (<bold>B</bold>) (Upper) Distribution of DAPs in HOT loci binned by the % of HOT loci they overlap with. (Lower) % of DAPs in the bins annotated as LLPS. Green points are the expected percentage values obtained by randomly shuffling the peaks in HOT loci 10 times. (<bold>C</bold>) Z-scores of ChIP-seq signal values of LLPS proteins and the rest of the DAPs in HOT loci. (<bold>D</bold>) % of the protein lengths predicted as IDRs (MobiDB) in LLPS proteins and the rest of the DAPs. (<bold>E</bold>) Enrichment of ChIP-seq peaks of RNA-binding proteins (RBP) and the rest of the DAPs. (<bold>F</bold>) Enrichment of FANTOM, PINTS, and CAGE regions in HOT, regular enhancers, and regular promoters. (<bold>G</bold>) Enrichment of eCLIP RBP-RNA interactions in HOT, exons, regular enhancers, and regular promoters. (<bold>E–G</bold>) Enrichment values are quantified as log2(fold-change) with ATAC-seq regions as a background. (<bold>C–E, G</bold>) Red dots represent the mean values of the boxplots.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-95170-fig9-v1.tif"/></fig><p>Next, we sought to quantify the RNA-related interactions in HOT loci. First, we used ENCODE’s set of ChIP-seq datasets extracted using RNA-binding proteins (RBP) and observed that RBPs are more enriched in HOT loci compared to the rest of the DAPs in terms of fold-increase using ATAC-seq regions as background (<xref ref-type="fig" rid="fig9">Figure 9E</xref>, 1.5 vs. 1.3 in log2(FC), p-value=0.04, t-test). Second, we quantified the level of transcription using FANTOM, PINTS (<xref ref-type="bibr" rid="bib74">Yao et al., 2022</xref>) (a modern tool for annotating eRNAs combining multiple types of RNA sequencing assays), and CAGE-seq peaks. We observed that all three types of annotations demonstrate high overrepresentation in HOT loci compared to regular promoters and enhancers by a factor of 2.7× on average (<xref ref-type="fig" rid="fig9">Figure 9F</xref>). Lastly, we used eCLIP datasets of 103 RBSs from the ENCODE Project and calculated the levels of RBP-RNA interactions. We observed that the difference in the levels of eCLIP signals in HOT loci and coding sequences are insignificant (1.31 vs. 1.4 in log2(FC), p-value=0.4, t-test), while in regular promoter and enhancer regions, the eCLIP signals are depleted compared to the ATAC-seq regions with the log2(FC) values of –0.1 and –0.05, respectively (p-value&lt;10<sup>–30</sup>, t-test), suggesting a strong RNA-related component in the composition of 3D medium surrounding the HOT loci.</p><p>All this data suggests a strong likelihood of involvement of transcriptional condensates in the mechanisms leading to the phenomena of HOT loci.</p></sec></sec><sec id="s3" sec-type="discussion"><title>Discussion</title><p>HOT loci have been noticed and studied in different species since the early years of the advent of the ChIP-seq datasets (<xref ref-type="bibr" rid="bib51">Roy et al., 2010</xref>; <xref ref-type="bibr" rid="bib40">Moorman et al., 2006</xref>; <xref ref-type="bibr" rid="bib19">Gerstein et al., 2010</xref>; <xref ref-type="bibr" rid="bib26">Kvon et al., 2012</xref>; <xref ref-type="bibr" rid="bib75">Yip et al., 2012</xref>; <xref ref-type="bibr" rid="bib73">Xie et al., 2013</xref>). Up until recently, most of the studies have extensively studied the reasons through which the ChIP-seq peaks appeared to be binding to HOT loci and characterized certain sequence features of the HOT loci which could enable elevated read mapping rates (<xref ref-type="bibr" rid="bib40">Moorman et al., 2006</xref>; <xref ref-type="bibr" rid="bib60">Teytelman et al., 2013</xref>; <xref ref-type="bibr" rid="bib71">Wreczycka et al., 2019</xref>). As the number of assayed DAPs in multiple human cell types and model organisms has increased, however, the assumption of the HOT loci being exceptional cases and results of false positives in ChIP-seq protocols have given way to the acceptance that the HOT loci, with exorbitant numbers of mapped TFBSs, are indeed hyperactive loci with distinct features characteristic of active regulatory regions (<xref ref-type="bibr" rid="bib48">Ramaker et al., 2020</xref>; <xref ref-type="bibr" rid="bib45">Partridge et al., 2020</xref>).</p><p>In this study, we studied the HOT loci in multiple complementary aspects to the previous works and expanded the scope of characterization extensively using the functional genomics datasets. We used the two most extensively characterized differentiated cell lines of the ENCODE Project: HepG2 and K562. We also included the H1-hESC human stem cells to study the activities of HOT loci during the embryonic stage. The number of assayed DAPs in these cell lines is far from complete (<xref ref-type="bibr" rid="bib27">Lambert et al., 2018</xref>), therefore it is important to note that as the sizes of the assayed DAP ChIP-seq datasets increase, our understanding of the mechanisms of HOT loci will certainly improve. However, the core principles can already be inferred using the currently available datasets. Previous studies have used different metrics to define the HOT loci. For example, <xref ref-type="bibr" rid="bib71">Wreczycka et al., 2019</xref>, used the 99th percentile of the density of TFBSs for a 500 bp sliding window, <xref ref-type="bibr" rid="bib48">Ramaker et al., 2020</xref>, used the window length of 2 kb and required &gt;25% of TFs to be mapped, <xref ref-type="bibr" rid="bib45">Partridge et al., 2020</xref>, used loci with &gt;70 chromatin-associated proteins in 2 kb window. These heterogeneous definitions, however, fail to appreciate that the histogram of loci binned by the number of harbored TFBSs represents an exponential distribution (<xref ref-type="fig" rid="fig1">Figure 1A</xref>). We, therefore, applied our analyses both to the binarily defined HOT and non-HOT loci, as well as to the overall spectrum of loci in the context of TFBS density. This approach allowed us to better understand the correlations of characteristics of loci with the TF activity. Noticeably, this approach showed us that the HOT loci have their propensities to engage in long-range chromatin contacts with other equally or more DAP-bound loci than less active ones, making it more clear that the HOT loci are located in 3D hubs and FIREs (<xref ref-type="fig" rid="fig3">Figure 3A</xref>).</p><p>Using the datasets generated in H1 we established that only &lt;10% of the HOT loci in two differentiated cell lines overlap with the HOT loci of stem cells. This points to the high tissue specificity of the HOT loci. Previous studies have also concluded that the HOT loci are not constitutive by nature, and are established in a dynamic manner after the differentiation (<xref ref-type="bibr" rid="bib8">Boyle et al., 2014</xref>).</p><p>Previous studies have carried out extensive mapping of the known binding motifs of TFs to the HOT loci and identified a small set of ‘anchor’ binding motifs of a few key tissue-specific TFs (<xref ref-type="bibr" rid="bib40">Moorman et al., 2006</xref>; <xref ref-type="bibr" rid="bib48">Ramaker et al., 2020</xref>), and proposed that perhaps these driver TFs initiated the formation of HOT loci, similar to how the pioneer factors function. Other studies have concluded that the vast majority of the peaks do not contain the corresponding motifs and that most of the mapped peaks represent indirect binding through TF-TF interactions (<xref ref-type="bibr" rid="bib48">Ramaker et al., 2020</xref>; <xref ref-type="bibr" rid="bib45">Partridge et al., 2020</xref>; <xref ref-type="bibr" rid="bib63">Vierstra et al., 2020</xref>; <xref ref-type="bibr" rid="bib69">White et al., 2021</xref>). We relied on these studies and focused on aspects of the HOT loci other than the quantification of known binding motifs of DAPs in HOT loci. Interestingly, the high prediction accuracy of our deep learning model is in agreement with the notion of the existence of shared motifs among the HOT loci but also implies that the indirectly bound loci also carry shared sequence features, perhaps other than the binding motifs or weak motifs which are not detected using the traditional PWM-based tools of motif detection.</p><p>Another model that has been increasingly attributed to the formation and maintenance of long-range 3D chromatin interactions involves phase-separated condensates (<xref ref-type="bibr" rid="bib41">Nair et al., 2019</xref>; <xref ref-type="bibr" rid="bib29">Lee et al., 2022</xref>; <xref ref-type="bibr" rid="bib16">Feric and Misteli, 2022</xref>; <xref ref-type="bibr" rid="bib2">Ahn et al., 2021</xref>). Some enhancers were shown to drive the formation of large chromosomal assemblies involving a high concentration of TFs (<xref ref-type="bibr" rid="bib41">Nair et al., 2019</xref>). In general, it has been increasingly appreciated that condensates ubiquitously attract and activate enhancers (<xref ref-type="bibr" rid="bib56">Shrinivas et al., 2019</xref>; <xref ref-type="bibr" rid="bib68">Wei et al., 2020</xref>; <xref ref-type="bibr" rid="bib7">Boija et al., 2018</xref>). The detection of condensates relies on low-throughput live-cell imaging methods such as FISH, which often involves only a few tagged molecules. Therefore, currently, to the best of our knowledge, there are no datasets of condensate formation with large numbers of molecules simultaneously that we could use to draw statistical conclusions. However, there is already an increasing body of research reporting on the characteristic hallmarks that the transcriptional condensates share (<xref ref-type="bibr" rid="bib43">Palacio and Taatjes, 2022</xref>; <xref ref-type="bibr" rid="bib38">Mitrea et al., 2022</xref>; <xref ref-type="bibr" rid="bib18">Gelder et al., 2024</xref>; <xref ref-type="bibr" rid="bib4">Bhat et al., 2021</xref>; <xref ref-type="bibr" rid="bib49">Rippe and Papantonis, 2021</xref>). We used those hallmarks as telltale signs and made a case for the likelihood of the HOT loci being sites with a high propensity of forming condensates. A condensate can start forming with only one bound TF and a cofactor, e.g. OCT4 and Mediator (<xref ref-type="bibr" rid="bib56">Shrinivas et al., 2019</xref>), which requires the presence of a strong binding motif of the condensate-initiating TF. Once the condensates of sufficient size form, the kinetic trap that it creates can facilitate the accumulation of a soup of DAPs, which then can undergo high-intensity protein-protein and protein-DNA and protein-RNA interactions, many constituents of which then get mapped to the involved DNA regions upon ChIP-seq experiments. This model can incorporate the seemingly contradictory conclusions of (a) the vast majority of DAPs lacking the binding motifs in HOT loci and (b) a high accuracy of sequence-based classification of HOT loci using the CNN models. It is important to note here that our proposed condensate model is a speculative hypothesis. Further experimental studies in the field are needed to confirm or reject it.</p><p>One of the main limitations of our study is the lack of higher-resolution TF-DNA interaction datasets such as CUT&amp;RUN, ChIP-exo, or single-cell versions of the assets used in this study. Furthermore, one of the hallmarks of condensates is the overrepresentation of certain structural motifs in LLPS proteins, which we did not pursue due to size limitations. Further studies addressing these topics hold promise to shed more light on the subject of HOT loci.</p></sec><sec id="s4" sec-type="methods"><title>Methods</title><sec id="s4-1"><title>Datasets</title><p>TF (DAP), histone modification, DHS ChIP-seq, and ATAC-seq datasets for HepG2, K562, H1-hESC cell lines were batch downloaded from the ENCODE Project (<xref ref-type="bibr" rid="bib66">Wang et al., 2013</xref>). For each DAP of each cell line, if there were multiple datasets, the one with the latest date was selected, prioritizing the ones with the least among the audit errors and warnings (<xref ref-type="supplementary-material" rid="supp1">Supplementary file 1, table S1</xref>). The GRCh37/hg19 assembly was used as a reference genome throughout the study. In those cases when ChIP-seq dataset was reported on GRCh38/hg38, the coordinates were converted to hg19 using liftOver. The phastCons evolutionary conservation scores generated from 46 vertebrate species, placental mammals, and primates. For comparing, averaged values of phastCons scores over the 400 bp loci were used. CpG islands, repeat elements, and GENCODE TSS annotations were all obtained from the UCSC genome browser database (<xref ref-type="bibr" rid="bib13">Davis et al., 2018</xref>). Transcribed enhancer regions (eRNAs) were obtained from the FANTOM database (<xref ref-type="bibr" rid="bib34">Lizio et al., 2019</xref>). Super-enhancer regions were obtained from <xref ref-type="bibr" rid="bib21">Hnisz et al., 2013</xref>.</p><p>Hi-C datasets were obtained from ENCODE Project. See Appendix 1 – Hi-C 3D chromatin analysis for a detailed description of Hi-C data analysis.</p><p>GC contents were calculated using the ‘nuc’' functionality of the bedtools program (<xref ref-type="bibr" rid="bib46">Quinlan and Hall, 2010</xref>). Gene expression data was obtained from the Roadmap Epigenomics Project. For analyzing the expression levels of target genes, the gene of the overlapping TSS was used for promoters, whereas for enhancers, the nearest genes were selected using the <italic>bedtools closest</italic> function. Tissue specificity metric <italic>tau</italic> scores for genes were downloaded from <xref ref-type="bibr" rid="bib44">Palmer et al., 2021</xref>.</p><p>LLPS protein annotations were obtained from CD-CODE website <ext-link ext-link-type="uri" xlink:href="https://cd-code.org">https://cd-code.org</ext-link>. Predicted intrinsically disordered region annotations of proteins were obtained from MobiDB website <ext-link ext-link-type="uri" xlink:href="https://mobidb.org">https://mobidb.org</ext-link>. RBP ChIP-seq datasets used in the study are in <xref ref-type="supplementary-material" rid="supp1">Supplementary file 1, table S6</xref>. eCLIP datasets used in the study are in <xref ref-type="supplementary-material" rid="supp1">Supplementary file 1, table S7</xref>. PINTS eRNA dataset was obtained from <ext-link ext-link-type="uri" xlink:href="https://pints.yulab.org">https://pints.yulab.org</ext-link>. CAGE datasets were downloaded from ENCODE (ENCFF184VBV, ENCFF246WDH, ENCFF933JJT) and merged.</p></sec><sec id="s4-2"><title>Definitions</title><p>The loci were divided into bins according to a two-part scale. The first part is on a linear scale from 1 to 5 (4 bins), the second part is on a natural logarithmic scale from 5 to the maximum number of DAPs bound to a single locus in that cell line (10 bins) (<xref ref-type="table" rid="table1">Table 1</xref>).</p><table-wrap id="table1" position="float"><label>Table 1.</label><caption><title>Schema of classifying loci according to the number of bound DNA-associated proteins (DAPs).</title><p>The initial 4 bins are loci bound by DAPs increasing linearly from 1 to 5 (gray fields). The remaining 10 bins are defined by edge values increasing on a logarithmic scale from 5 to the maximum number of available DAPs in each cell line (orange and red fields) using the Numpy formula np.logspace(np.log10(5), np.log10(max_tfs), 11, dtype = int). HOT loci correspond to the last 5 bin edges (red fields).</p></caption><table frame="hsides" rules="groups"><thead><tr><th align="left" valign="bottom"/><th align="left" valign="bottom" colspan="15">Bin edges (n=15)</th></tr></thead><tbody><tr><td align="left" valign="bottom">HepG2</td><td style="background-color: #E6E6E6;">1</td><td style="background-color: #E6E6E6;">2</td><td style="background-color: #E6E6E6;">3</td><td style="background-color: #E6E6E6;">4</td><td style="background-color: #E6E6E6;">5</td><td style="background-color: #FFB74D;">7</td><td style="background-color: #FFB74D;">12</td><td style="background-color: #FFB74D;">19</td><td style="background-color: #FFB74D;">31</td><td style="background-color: #FFB74D;">48</td><td style="background-color: #E57373;">77</td><td style="background-color: #E57373;">122</td><td style="background-color: #E57373;">192</td><td style="background-color: #E57373;">304</td><td style="background-color: #E57373;">480</td></tr><tr><td align="left" valign="bottom">K562</td><td style="background-color: #E6E6E6;">1</td><td style="background-color: #E6E6E6;">2</td><td style="background-color: #E6E6E6;">3</td><td style="background-color: #E6E6E6;">4</td><td style="background-color: #E6E6E6;">5</td><td style="background-color: #FFB74D;">7</td><td style="background-color: #FFB74D;">11</td><td style="background-color: #FFB74D;">16</td><td style="background-color: #FFB74D;">24</td><td style="background-color: #FFB74D;">37</td><td style="background-color: #E57373;">55</td><td style="background-color: #E57373;">82</td><td style="background-color: #E57373;">123</td><td style="background-color: #E57373;">184</td><td style="background-color: #E57373;">275</td></tr><tr><td align="left" valign="bottom">H1</td><td style="background-color: #E6E6E6;">1</td><td style="background-color: #E6E6E6;">2</td><td style="background-color: #E6E6E6;">3</td><td style="background-color: #E6E6E6;">4</td><td style="background-color: #E6E6E6;">5</td><td style="background-color: #FFB74D;">6</td><td style="background-color: #FFB74D;">7</td><td style="background-color: #FFB74D;">8</td><td style="background-color: #FFB74D;">10</td><td style="background-color: #FFB74D;">12</td><td style="background-color: #E57373;">15</td><td style="background-color: #E57373;">18</td><td style="background-color: #E57373;">22</td><td style="background-color: #E57373;">26</td><td style="background-color: #E57373;">32</td></tr><tr><td align="left" valign="bottom"/><td align="left" valign="bottom" colspan="5"><bold>Linear growth (n=4</bold>)</td><td align="left" valign="bottom" colspan="10"><bold>Logarithmic growth (n=10</bold>)</td></tr></tbody></table></table-wrap><p>We considered an average TFBS to be 8 bp long (<xref ref-type="bibr" rid="bib64">Vinson et al., 2011</xref>; <xref ref-type="bibr" rid="bib72">Wunderlich and Mirny, 2009</xref>). Given that we analyzed the loci in 400 bp, we reasoned that, theoretically, there can be at most 50 simultaneous binding events in the locus (8×50 = 400). Therefore, we considered the bins containing &gt;50 DAPs in K562 as HOT loci, which meant the last four bins in <xref ref-type="table" rid="table1">Table 1</xref>. The reason we chose K562 for setting the threshold was the fact that K562 is the lesser of the two most TF ChIP-seq abundant cell lines. So, the corresponding threshold number for HepG2 is &gt;77 TFs.</p><p>These nominal numbers are used in cases when the distributions are displayed for individual cell lines (such as <xref ref-type="fig" rid="fig1">Figure 1A</xref> and <xref ref-type="fig" rid="fig1s1">Figure 1—figure supplement 1</xref>). When the figures display the distributions for two cell lines in a joint manner (such as <xref ref-type="fig" rid="fig3">Figure 3A and B</xref>), the edges are converted to the average percentages of the overall scale lengths for each cell line.</p><p><italic>Regular enhancers</italic> were defined as central 400 bp regions of DHS which overlap H3K27ac histone modification regions with promoter and exons removed from them.</p><p><italic>Promoters</italic> were defined as 1.5 kb upstream and 500 bp downstream regions of the canonical and alternative TSS coordinates were extracted from the knownGenes.txt table obtained from UCSC Genome Browser.</p><p>All the genomic arithmetic operations were done using the <italic>bedtools</italic> program (<xref ref-type="bibr" rid="bib46">Quinlan and Hall, 2010</xref>). Figures were generated using Matplotlib (<xref ref-type="bibr" rid="bib24">Hunter, 2007</xref>) and Seaborn (<xref ref-type="bibr" rid="bib67">Waskom, 2021</xref>) packages. Statistical and numerical analyses were done using the pandas, <italic>NumPy</italic>, <italic>SciPy,</italic> and <italic>sklearn</italic> packages (<xref ref-type="bibr" rid="bib65">Virtanen et al., 2020</xref>) in <italic>Python</italic> programming language. Genomic repeat regions were extracted from <italic>RepeatMasker</italic> table obtained from <ext-link ext-link-type="uri" xlink:href="http://www.repeatmasker.org/">http://www.repeatmasker.org/</ext-link>. CpG islands were extracted from <italic>cpgIslandExt</italic> table obtained from the UCSC Genome Browser. Protein-protein interaction network information was obtained using the <ext-link ext-link-type="uri" xlink:href="https://string-db.org">https://string-db.org</ext-link> web interface (<xref ref-type="bibr" rid="bib59">Szklarczyk et al., 2019</xref>).</p></sec><sec id="s4-3"><title>Statistical analyses</title><p>All the statistical significance analyses were done using the <italic>SciPy</italic> package. Statistical significance of genomic region overlaps was calculated using the ‘<italic>bedtools fisher</italic>’ command. The p-values too small to be represented by the command line output were represented as &lt;10<sup>–100</sup>.</p><p>Correlation values with the number of bound TFs were calculated using the average of the value for the bins, and the midpoint numbers of the edges of each bin.</p><p>For calculating the statistical significance, we used the non-parametric Mann-Whitney U test when the compared data points are non-linearly correlated and multi-modal. When the data distributions are bell-curve shaped, the Student’s t-test was used.</p></sec><sec id="s4-4"><title>GWAS analysis</title><p>NHGRI-EBI GWAS database variants were grouped according to their traits (dataset e0_r2022-11-29). For each GWAS SNP, LD SNPs with r2&gt;0.8 were added using the <italic>plink v1.9</italic> (<xref ref-type="bibr" rid="bib9">Chang et al., 2015</xref>) program using the parameters <italic><monospace>--ld-window-r2 0.8</monospace> <monospace>--ld-window-kb 100</monospace> <monospace>--ld-window 1000000</monospace></italic>. Enrichments of GWAS-trait SNPs were calculated as the ratios of densities of SNPs in each class of regions (e.g. HOT enhancers, HOT promoters) to either that of the regular enhancers or the DHS regions. Statistical significance of enrichment was calculated using the binomial test. FDR values were calculated using the Bonferroni correction.</p></sec><sec id="s4-5"><title>Sequence classification analysis</title><p>Classification tasks were constructed in a binary classification setup. The control regions were used from: (a) randomly selected (10× the size of the HOT loci) merged DHS regions from all the available datasets from Roadmap Epigenomic Project, (b) all of the promoter regions as defined above, (c) regular enhancers as defined above, with the HOT loci subtracted (see Appendix 1 – Classification datasets for details).</p><sec id="s4-5-1"><title>Sequence-based classification (CNN)</title><p>Sequences were converted to one-hot encoding and a CNN was trained using each of the control regions as negative set. The model was built using <italic>tensorflow v2.3.1</italic> (<xref ref-type="bibr" rid="bib1">Abadi et al., 2016</xref>) and trained on NVIDIA k80 GPUs (see Appendix 1 – Sequence-based classification for details).</p></sec><sec id="s4-5-2"><title>Sequence-based classification (SVM)</title><p>SVM models were trained using the LS-GKM package (<xref ref-type="bibr" rid="bib28">Lee, 2016</xref>) (see Appendix 1 – Sequence-based classification for details).</p></sec><sec id="s4-5-3"><title>Feature-based classification</title><p>Sequences were represented in terms of GC, CpG, GpC contents and overlap percentages with annotated CpG islands. SVM classifiers were trained using these sequence features (see Appendix 1 – Feature-based classification for details).</p></sec></sec><sec id="s4-6"><title>Variant analysis</title><p>Common SNPs and INDELs were extracted from the <italic>gnomAD r2.1.1</italic> dataset (<xref ref-type="bibr" rid="bib25">Karczewski et al., 2020</xref>). Variants with PASS filter value and MAF&gt;5% were selected using the “view -f PASS -i 'MAF[0]&gt;0.05'” options of <italic>bcftools</italic> program (<xref ref-type="bibr" rid="bib31">Li, 2011</xref>). Loss-of-function variants were downloaded from the <italic>gnomAD</italic> website under the option ‘all homozygous LoF curation’ section of v2.1.1 database. raQTLs were downloaded from <ext-link ext-link-type="uri" xlink:href="https://sure.nki.nl">https://sure.nki.nl</ext-link> (<xref ref-type="bibr" rid="bib62">van Arensbergen et al., 2019</xref>). Liver and blood eQTLs were extracted from the GTEx v8 dataset (<ext-link ext-link-type="uri" xlink:href="https://www.gtexportal.org/home/datasets">https://www.gtexportal.org/home/datasets</ext-link>). Liver caQTLs were obtained from the supplementary material of <xref ref-type="bibr" rid="bib11">Currin et al., 2021</xref>. NHGRI-EBI GWAS database variants were grouped according to their traits (dataset e0_r2022-11-29). For each GWAS SNP, LD SNPs with r2&gt;0.8 were added using the <italic>plink v1.9</italic> program using the parameters ‘<italic><monospace>--ld-window-r2 0.8</monospace> <monospace>--ld-window-kb 100</monospace> <monospace>--ld-window 1000000</monospace>’</italic>. Enrichments of GWAS-trait SNPs were calculated as the ratios of densities of SNPs in each class of regions (e.g. HOT enhancers, HOT promoters) to either that of the regular enhancers or the DHS regions. The statistical significance of enrichment was calculated using the binomial test. FDR values were calculated using the Bonferroni correction.</p></sec></sec></body><back><sec sec-type="additional-information" id="s5"><title>Additional information</title><fn-group content-type="competing-interest"><title>Competing interests</title><fn fn-type="COI-statement" id="conf1"><p>No competing interests declared</p></fn></fn-group><fn-group content-type="author-contribution"><title>Author contributions</title><fn fn-type="con" id="con1"><p>Conceptualization, Visualization, Methodology, Writing – original draft, Data curation, Formal analysis, Investigation, Resources, Software, Validation</p></fn><fn fn-type="con" id="con2"><p>Conceptualization, Formal analysis, Supervision, Project administration, Writing – review and editing, Investigation, Validation</p></fn></fn-group></sec><sec sec-type="supplementary-material" id="s6"><title>Additional files</title><supplementary-material id="mdar"><label>MDAR checklist</label><media xlink:href="elife-95170-mdarchecklist1-v1.docx" mimetype="application" mime-subtype="docx"/></supplementary-material><supplementary-material id="supp1"><label>Supplementary file 1.</label><caption><title>Supplementary tables.</title><p>Columns are explained in each sheet. S1: List of ENCODE ChIP-seq datasets used in the study. S2: Coordinates of high-occupancy target (HOT) loci defined in three cell lines. S3: List of DNA-associated proteins (DAPs) clustered into four groups and their PPI enrichment summary statistics. S4: List of ultraconserved regions overlapping with HOT loci. S5: Comparison of HOT loci defined using ENCODE vs. Roadmap Epigenome Project datasets. S6: List of ChIP-seq datasets of RNA-binding protein used in the study. S7: List of eCLIP datasets of RNA-binding proteins used in the study.</p></caption><media xlink:href="elife-95170-supp1-v1.xlsx" mimetype="application" mime-subtype="xlsx"/></supplementary-material></sec><sec sec-type="data-availability" id="s7"><title>Data availability</title><p>All the used and produced data presented in this manuscript are deposited in <ext-link ext-link-type="uri" xlink:href="https://zenodo.org/records/13271790">Zenodo</ext-link>. The codebase used for generating the results presented in this manuscript is available at <ext-link ext-link-type="uri" xlink:href="https://github.com/okurman/HOT">GitHub</ext-link>, copy archived at <xref ref-type="bibr" rid="bib23">Hudaiberdiev, 2024</xref>.</p><p>The following dataset was generated:</p><p><element-citation publication-type="data" specific-use="isSupplementedBy" id="dataset1"><person-group person-group-type="author"><name><surname>Hudaiberdiav</surname><given-names>S</given-names></name><name><surname>Ovcharenko</surname><given-names>I</given-names></name></person-group><year iso-8601-date="2024">2024</year><data-title>Functional characteristics and computational model of abundant hyperactive loci in the human genome</data-title><source>Zenodo</source><pub-id pub-id-type="doi">10.5281/zenodo.7845120</pub-id></element-citation></p></sec><ack id="ack"><title>Acknowledgements</title><p>This work utilized the computational resources of the NIH HPC Biowulf cluster (<ext-link ext-link-type="uri" xlink:href="http://hpc.nih.gov">http://hpc.nih.gov</ext-link>). This research was supported by the Intramural Research Program of the National Library of Medicine, National Institutes of Health.</p></ack><ref-list><title>References</title><ref id="bib1"><element-citation publication-type="preprint"><person-group person-group-type="author"><name><surname>Abadi</surname><given-names>M</given-names></name><name><surname>Agarwal</surname><given-names>A</given-names></name><name><surname>Barham</surname><given-names>P</given-names></name><name><surname>Brevdo</surname><given-names>E</given-names></name><name><surname>Chen</surname><given-names>Z</given-names></name><name><surname>Citro</surname><given-names>C</given-names></name><name><surname>Corrado</surname><given-names>GS</given-names></name><name><surname>Davis</surname><given-names>A</given-names></name><name><surname>Dean</surname><given-names>J</given-names></name><name><surname>Devin</surname><given-names>M</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>TensorFlow: Large-Scale Machine Learning on Heterogeneous Distributed Systems</article-title><source>arXiv</source><pub-id pub-id-type="doi">10.48550/arXiv.1603.04467</pub-id></element-citation></ref><ref id="bib2"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ahn</surname><given-names>JH</given-names></name><name><surname>Davis</surname><given-names>ES</given-names></name><name><surname>Daugird</surname><given-names>TA</given-names></name><name><surname>Zhao</surname><given-names>S</given-names></name><name><surname>Quiroga</surname><given-names>IY</given-names></name><name><surname>Uryu</surname><given-names>H</given-names></name><name><surname>Li</surname><given-names>J</given-names></name><name><surname>Storey</surname><given-names>AJ</given-names></name><name><surname>Tsai</surname><given-names>YH</given-names></name><name><surname>Keeley</surname><given-names>DP</given-names></name><name><surname>Mackintosh</surname><given-names>SG</given-names></name><name><surname>Edmondson</surname><given-names>RD</given-names></name><name><surname>Byrum</surname><given-names>SD</given-names></name><name><surname>Cai</surname><given-names>L</given-names></name><name><surname>Tackett</surname><given-names>AJ</given-names></name><name><surname>Zheng</surname><given-names>D</given-names></name><name><surname>Legant</surname><given-names>WR</given-names></name><name><surname>Phanstiel</surname><given-names>DH</given-names></name><name><surname>Wang</surname><given-names>GG</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>Phase separation drives aberrant chromatin looping and cancer development</article-title><source>Nature</source><volume>595</volume><fpage>591</fpage><lpage>595</lpage><pub-id pub-id-type="doi">10.1038/s41586-021-03662-5</pub-id><pub-id pub-id-type="pmid">34163069</pub-id></element-citation></ref><ref id="bib3"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Arnosti</surname><given-names>DN</given-names></name><name><surname>Kulkarni</surname><given-names>MM</given-names></name></person-group><year iso-8601-date="2005">2005</year><article-title>Transcriptional enhancers: intelligent enhanceosomes or flexible billboards?</article-title><source>Journal of Cellular Biochemistry</source><volume>94</volume><fpage>890</fpage><lpage>898</lpage><pub-id pub-id-type="doi">10.1002/jcb.20352</pub-id><pub-id pub-id-type="pmid">15696541</pub-id></element-citation></ref><ref id="bib4"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Bhat</surname><given-names>P</given-names></name><name><surname>Honson</surname><given-names>D</given-names></name><name><surname>Guttman</surname><given-names>M</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>Nuclear compartmentalization as a mechanism of quantitative control of gene expression</article-title><source>Nature Reviews. Molecular Cell Biology</source><volume>22</volume><fpage>653</fpage><lpage>670</lpage><pub-id pub-id-type="doi">10.1038/s41580-021-00387-1</pub-id><pub-id pub-id-type="pmid">34341548</pub-id></element-citation></ref><ref id="bib5"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Bhattacharyya</surname><given-names>S</given-names></name><name><surname>Chandra</surname><given-names>V</given-names></name><name><surname>Vijayanand</surname><given-names>P</given-names></name><name><surname>Ay</surname><given-names>F</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Identification of significant chromatin contacts from HiChIP data by FitHiChIP</article-title><source>Nature Communications</source><volume>10</volume><elocation-id>4221</elocation-id><pub-id pub-id-type="doi">10.1038/s41467-019-11950-y</pub-id><pub-id pub-id-type="pmid">31530818</pub-id></element-citation></ref><ref id="bib6"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Biel</surname><given-names>A</given-names></name><name><surname>Castanza</surname><given-names>AS</given-names></name><name><surname>Rutherford</surname><given-names>R</given-names></name><name><surname>Fair</surname><given-names>SR</given-names></name><name><surname>Chifamba</surname><given-names>L</given-names></name><name><surname>Wester</surname><given-names>JC</given-names></name><name><surname>Hester</surname><given-names>ME</given-names></name><name><surname>Hevner</surname><given-names>RF</given-names></name></person-group><year iso-8601-date="2022">2022</year><article-title>AUTS2 syndrome: molecular mechanisms and model systems</article-title><source>Frontiers in Molecular Neuroscience</source><volume>15</volume><elocation-id>858582</elocation-id><pub-id pub-id-type="doi">10.3389/fnmol.2022.858582</pub-id><pub-id pub-id-type="pmid">35431798</pub-id></element-citation></ref><ref id="bib7"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Boija</surname><given-names>A</given-names></name><name><surname>Klein</surname><given-names>IA</given-names></name><name><surname>Sabari</surname><given-names>BR</given-names></name><name><surname>Dall’Agnese</surname><given-names>A</given-names></name><name><surname>Coffey</surname><given-names>EL</given-names></name><name><surname>Zamudio</surname><given-names>AV</given-names></name><name><surname>Li</surname><given-names>CH</given-names></name><name><surname>Shrinivas</surname><given-names>K</given-names></name><name><surname>Manteiga</surname><given-names>JC</given-names></name><name><surname>Hannett</surname><given-names>NM</given-names></name><name><surname>Abraham</surname><given-names>BJ</given-names></name><name><surname>Afeyan</surname><given-names>LK</given-names></name><name><surname>Guo</surname><given-names>YE</given-names></name><name><surname>Rimel</surname><given-names>JK</given-names></name><name><surname>Fant</surname><given-names>CB</given-names></name><name><surname>Schuijers</surname><given-names>J</given-names></name><name><surname>Lee</surname><given-names>TI</given-names></name><name><surname>Taatjes</surname><given-names>DJ</given-names></name><name><surname>Young</surname><given-names>RA</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Transcription factors activate genes through the phase-separation capacity of their activation domains</article-title><source>Cell</source><volume>175</volume><fpage>1842</fpage><lpage>1855</lpage><pub-id pub-id-type="doi">10.1016/j.cell.2018.10.042</pub-id><pub-id pub-id-type="pmid">30449618</pub-id></element-citation></ref><ref id="bib8"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Boyle</surname><given-names>AP</given-names></name><name><surname>Araya</surname><given-names>CL</given-names></name><name><surname>Brdlik</surname><given-names>C</given-names></name><name><surname>Cayting</surname><given-names>P</given-names></name><name><surname>Cheng</surname><given-names>C</given-names></name><name><surname>Cheng</surname><given-names>Y</given-names></name><name><surname>Gardner</surname><given-names>K</given-names></name><name><surname>Hillier</surname><given-names>LW</given-names></name><name><surname>Janette</surname><given-names>J</given-names></name><name><surname>Jiang</surname><given-names>L</given-names></name><name><surname>Kasper</surname><given-names>D</given-names></name><name><surname>Kawli</surname><given-names>T</given-names></name><name><surname>Kheradpour</surname><given-names>P</given-names></name><name><surname>Kundaje</surname><given-names>A</given-names></name><name><surname>Li</surname><given-names>JJ</given-names></name><name><surname>Ma</surname><given-names>L</given-names></name><name><surname>Niu</surname><given-names>W</given-names></name><name><surname>Rehm</surname><given-names>EJ</given-names></name><name><surname>Rozowsky</surname><given-names>J</given-names></name><name><surname>Slattery</surname><given-names>M</given-names></name><name><surname>Spokony</surname><given-names>R</given-names></name><name><surname>Terrell</surname><given-names>R</given-names></name><name><surname>Vafeados</surname><given-names>D</given-names></name><name><surname>Wang</surname><given-names>D</given-names></name><name><surname>Weisdepp</surname><given-names>P</given-names></name><name><surname>Wu</surname><given-names>YC</given-names></name><name><surname>Xie</surname><given-names>D</given-names></name><name><surname>Yan</surname><given-names>KK</given-names></name><name><surname>Feingold</surname><given-names>EA</given-names></name><name><surname>Good</surname><given-names>PJ</given-names></name><name><surname>Pazin</surname><given-names>MJ</given-names></name><name><surname>Huang</surname><given-names>H</given-names></name><name><surname>Bickel</surname><given-names>PJ</given-names></name><name><surname>Brenner</surname><given-names>SE</given-names></name><name><surname>Reinke</surname><given-names>V</given-names></name><name><surname>Waterston</surname><given-names>RH</given-names></name><name><surname>Gerstein</surname><given-names>M</given-names></name><name><surname>White</surname><given-names>KP</given-names></name><name><surname>Kellis</surname><given-names>M</given-names></name><name><surname>Snyder</surname><given-names>M</given-names></name></person-group><year iso-8601-date="2014">2014</year><article-title>Comparative analysis of regulatory information and circuits across distant species</article-title><source>Nature</source><volume>512</volume><fpage>453</fpage><lpage>456</lpage><pub-id pub-id-type="doi">10.1038/nature13668</pub-id><pub-id pub-id-type="pmid">25164757</pub-id></element-citation></ref><ref id="bib9"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Chang</surname><given-names>CC</given-names></name><name><surname>Chow</surname><given-names>CC</given-names></name><name><surname>Tellier</surname><given-names>LC</given-names></name><name><surname>Vattikuti</surname><given-names>S</given-names></name><name><surname>Purcell</surname><given-names>SM</given-names></name><name><surname>Lee</surname><given-names>JJ</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>Second-generation PLINK: rising to the challenge of larger and richer datasets</article-title><source>GigaScience</source><volume>4</volume><elocation-id>7</elocation-id><pub-id pub-id-type="doi">10.1186/s13742-015-0047-8</pub-id><pub-id pub-id-type="pmid">25722852</pub-id></element-citation></ref><ref id="bib10"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Consortium</surname><given-names>G</given-names></name></person-group><year iso-8601-date="2015">2015</year><article-title>Human genomics: the genotype-tissue expression (gtex) pilot analysis: multitissue gene regulation in humans</article-title><source>Science</source><volume>348</volume><fpage>648</fpage><lpage>660</lpage><pub-id pub-id-type="doi">10.1126/science.1262110</pub-id></element-citation></ref><ref id="bib11"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Currin</surname><given-names>KW</given-names></name><name><surname>Erdos</surname><given-names>MR</given-names></name><name><surname>Narisu</surname><given-names>N</given-names></name><name><surname>Rai</surname><given-names>V</given-names></name><name><surname>Vadlamudi</surname><given-names>S</given-names></name><name><surname>Perrin</surname><given-names>HJ</given-names></name><name><surname>Idol</surname><given-names>JR</given-names></name><name><surname>Yan</surname><given-names>T</given-names></name><name><surname>Albanus</surname><given-names>RD</given-names></name><name><surname>Broadaway</surname><given-names>KA</given-names></name><name><surname>Etheridge</surname><given-names>AS</given-names></name><name><surname>Bonnycastle</surname><given-names>LL</given-names></name><name><surname>Orchard</surname><given-names>P</given-names></name><name><surname>Didion</surname><given-names>JP</given-names></name><name><surname>Chaudhry</surname><given-names>AS</given-names></name><collab>NISC Comparative Sequencing Program</collab><name><surname>Innocenti</surname><given-names>F</given-names></name><name><surname>Schuetz</surname><given-names>EG</given-names></name><name><surname>Scott</surname><given-names>LJ</given-names></name><name><surname>Parker</surname><given-names>SCJ</given-names></name><name><surname>Collins</surname><given-names>FS</given-names></name><name><surname>Mohlke</surname><given-names>KL</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>Genetic effects on liver chromatin accessibility identify disease regulatory variants</article-title><source>American Journal of Human Genetics</source><volume>108</volume><fpage>1169</fpage><lpage>1189</lpage><pub-id pub-id-type="doi">10.1016/j.ajhg.2021.05.001</pub-id><pub-id pub-id-type="pmid">34038741</pub-id></element-citation></ref><ref id="bib12"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Daigle</surname><given-names>TL</given-names></name><name><surname>Madisen</surname><given-names>L</given-names></name><name><surname>Hage</surname><given-names>TA</given-names></name><name><surname>Valley</surname><given-names>MT</given-names></name><name><surname>Knoblich</surname><given-names>U</given-names></name><name><surname>Larsen</surname><given-names>RS</given-names></name><name><surname>Takeno</surname><given-names>MM</given-names></name><name><surname>Huang</surname><given-names>L</given-names></name><name><surname>Gu</surname><given-names>H</given-names></name><name><surname>Larsen</surname><given-names>R</given-names></name><name><surname>Mills</surname><given-names>M</given-names></name><name><surname>Bosma-Moody</surname><given-names>A</given-names></name><name><surname>Siverts</surname><given-names>LA</given-names></name><name><surname>Walker</surname><given-names>M</given-names></name><name><surname>Graybuck</surname><given-names>LT</given-names></name><name><surname>Yao</surname><given-names>Z</given-names></name><name><surname>Fong</surname><given-names>O</given-names></name><name><surname>Nguyen</surname><given-names>TN</given-names></name><name><surname>Garren</surname><given-names>E</given-names></name><name><surname>Lenz</surname><given-names>GH</given-names></name><name><surname>Chavarha</surname><given-names>M</given-names></name><name><surname>Pendergraft</surname><given-names>J</given-names></name><name><surname>Harrington</surname><given-names>J</given-names></name><name><surname>Hirokawa</surname><given-names>KE</given-names></name><name><surname>Harris</surname><given-names>JA</given-names></name><name><surname>Nicovich</surname><given-names>PR</given-names></name><name><surname>McGraw</surname><given-names>MJ</given-names></name><name><surname>Ollerenshaw</surname><given-names>DR</given-names></name><name><surname>Smith</surname><given-names>KA</given-names></name><name><surname>Baker</surname><given-names>CA</given-names></name><name><surname>Ting</surname><given-names>JT</given-names></name><name><surname>Sunkin</surname><given-names>SM</given-names></name><name><surname>Lecoq</surname><given-names>J</given-names></name><name><surname>Lin</surname><given-names>MZ</given-names></name><name><surname>Boyden</surname><given-names>ES</given-names></name><name><surname>Murphy</surname><given-names>GJ</given-names></name><name><surname>da Costa</surname><given-names>NM</given-names></name><name><surname>Waters</surname><given-names>J</given-names></name><name><surname>Li</surname><given-names>L</given-names></name><name><surname>Tasic</surname><given-names>B</given-names></name><name><surname>Zeng</surname><given-names>H</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>A suite of transgenic driver and reporter mouse lines with enhanced brain-cell-type targeting and functionality</article-title><source>Cell</source><volume>174</volume><fpage>465</fpage><lpage>480</lpage><pub-id pub-id-type="doi">10.1016/j.cell.2018.06.035</pub-id><pub-id pub-id-type="pmid">30007418</pub-id></element-citation></ref><ref id="bib13"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Davis</surname><given-names>CA</given-names></name><name><surname>Hitz</surname><given-names>BC</given-names></name><name><surname>Sloan</surname><given-names>CA</given-names></name><name><surname>Chan</surname><given-names>ET</given-names></name><name><surname>Davidson</surname><given-names>JM</given-names></name><name><surname>Gabdank</surname><given-names>I</given-names></name><name><surname>Hilton</surname><given-names>JA</given-names></name><name><surname>Jain</surname><given-names>K</given-names></name><name><surname>Baymuradov</surname><given-names>UK</given-names></name><name><surname>Narayanan</surname><given-names>AK</given-names></name><name><surname>Onate</surname><given-names>KC</given-names></name><name><surname>Graham</surname><given-names>K</given-names></name><name><surname>Miyasato</surname><given-names>SR</given-names></name><name><surname>Dreszer</surname><given-names>TR</given-names></name><name><surname>Strattan</surname><given-names>JS</given-names></name><name><surname>Jolanki</surname><given-names>O</given-names></name><name><surname>Tanaka</surname><given-names>FY</given-names></name><name><surname>Cherry</surname><given-names>JM</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>The Encyclopedia of DNA elements (ENCODE): data portal update</article-title><source>Nucleic Acids Research</source><volume>46</volume><fpage>D794</fpage><lpage>D801</lpage><pub-id pub-id-type="doi">10.1093/nar/gkx1081</pub-id><pub-id pub-id-type="pmid">29126249</pub-id></element-citation></ref><ref id="bib14"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Deaton</surname><given-names>AM</given-names></name><name><surname>Bird</surname><given-names>A</given-names></name></person-group><year iso-8601-date="2011">2011</year><article-title>CpG islands and the regulation of transcription</article-title><source>Genes &amp; Development</source><volume>25</volume><fpage>1010</fpage><lpage>1022</lpage><pub-id pub-id-type="doi">10.1101/gad.2037511</pub-id><pub-id pub-id-type="pmid">21576262</pub-id></element-citation></ref><ref id="bib15"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Domené</surname><given-names>S</given-names></name><name><surname>Bumaschny</surname><given-names>VF</given-names></name><name><surname>de Souza</surname><given-names>FSJ</given-names></name><name><surname>Franchini</surname><given-names>LF</given-names></name><name><surname>Nasif</surname><given-names>S</given-names></name><name><surname>Low</surname><given-names>MJ</given-names></name><name><surname>Rubinstein</surname><given-names>M</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>Enhancer turnover and conserved regulatory function in vertebrate evolution</article-title><source>Philosophical Transactions of the Royal Society of London. Series B, Biological Sciences</source><volume>368</volume><elocation-id>20130027</elocation-id><pub-id pub-id-type="doi">10.1098/rstb.2013.0027</pub-id><pub-id pub-id-type="pmid">24218639</pub-id></element-citation></ref><ref id="bib16"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Feric</surname><given-names>M</given-names></name><name><surname>Misteli</surname><given-names>T</given-names></name></person-group><year iso-8601-date="2022">2022</year><article-title>Function moves biomolecular condensates in phase space</article-title><source>BioEssays</source><volume>44</volume><elocation-id>e2200001</elocation-id><pub-id pub-id-type="doi">10.1002/bies.202200001</pub-id><pub-id pub-id-type="pmid">35243657</pub-id></element-citation></ref><ref id="bib17"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Forsberg</surname><given-names>M</given-names></name><name><surname>Westin</surname><given-names>G</given-names></name></person-group><year iso-8601-date="1991">1991</year><article-title>Enhancer activation by a single type of transcription factor shows cell type dependence</article-title><source>The EMBO Journal</source><volume>10</volume><fpage>2543</fpage><lpage>2551</lpage><pub-id pub-id-type="doi">10.1002/j.1460-2075.1991.tb07794.x</pub-id><pub-id pub-id-type="pmid">1714381</pub-id></element-citation></ref><ref id="bib18"><element-citation publication-type="preprint"><person-group person-group-type="author"><name><surname>Gelder</surname><given-names>KL</given-names></name><name><surname>Carruthers</surname><given-names>NA</given-names></name><name><surname>Ball</surname><given-names>S</given-names></name><name><surname>Dunning</surname><given-names>M</given-names></name><name><surname>Craggs</surname><given-names>TD</given-names></name><name><surname>Twelvetrees</surname><given-names>AE</given-names></name><name><surname>Bose</surname><given-names>DA</given-names></name></person-group><year iso-8601-date="2024">2024</year><article-title>Cooperation between Intrinsically Disordered Regions Regulates CBP Condensate Behaviour</article-title><source>bioRxiv</source><pub-id pub-id-type="doi">10.1101/2024.06.04.597392</pub-id></element-citation></ref><ref id="bib19"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Gerstein</surname><given-names>MB</given-names></name><name><surname>Lu</surname><given-names>ZJ</given-names></name><name><surname>Van Nostrand</surname><given-names>EL</given-names></name><name><surname>Cheng</surname><given-names>C</given-names></name><name><surname>Arshinoff</surname><given-names>BI</given-names></name><name><surname>Liu</surname><given-names>T</given-names></name><name><surname>Yip</surname><given-names>KY</given-names></name><name><surname>Robilotto</surname><given-names>R</given-names></name><name><surname>Rechtsteiner</surname><given-names>A</given-names></name><name><surname>Ikegami</surname><given-names>K</given-names></name><name><surname>Alves</surname><given-names>P</given-names></name><name><surname>Chateigner</surname><given-names>A</given-names></name><name><surname>Perry</surname><given-names>M</given-names></name><name><surname>Morris</surname><given-names>M</given-names></name><name><surname>Auerbach</surname><given-names>RK</given-names></name><name><surname>Feng</surname><given-names>X</given-names></name><name><surname>Leng</surname><given-names>J</given-names></name><name><surname>Vielle</surname><given-names>A</given-names></name><name><surname>Niu</surname><given-names>W</given-names></name><name><surname>Rhrissorrakrai</surname><given-names>K</given-names></name><name><surname>Agarwal</surname><given-names>A</given-names></name><name><surname>Alexander</surname><given-names>RP</given-names></name><name><surname>Barber</surname><given-names>G</given-names></name><name><surname>Brdlik</surname><given-names>CM</given-names></name><name><surname>Brennan</surname><given-names>J</given-names></name><name><surname>Brouillet</surname><given-names>JJ</given-names></name><name><surname>Carr</surname><given-names>A</given-names></name><name><surname>Cheung</surname><given-names>MS</given-names></name><name><surname>Clawson</surname><given-names>H</given-names></name><name><surname>Contrino</surname><given-names>S</given-names></name><name><surname>Dannenberg</surname><given-names>LO</given-names></name><name><surname>Dernburg</surname><given-names>AF</given-names></name><name><surname>Desai</surname><given-names>A</given-names></name><name><surname>Dick</surname><given-names>L</given-names></name><name><surname>Dosé</surname><given-names>AC</given-names></name><name><surname>Du</surname><given-names>J</given-names></name><name><surname>Egelhofer</surname><given-names>T</given-names></name><name><surname>Ercan</surname><given-names>S</given-names></name><name><surname>Euskirchen</surname><given-names>G</given-names></name><name><surname>Ewing</surname><given-names>B</given-names></name><name><surname>Feingold</surname><given-names>EA</given-names></name><name><surname>Gassmann</surname><given-names>R</given-names></name><name><surname>Good</surname><given-names>PJ</given-names></name><name><surname>Green</surname><given-names>P</given-names></name><name><surname>Gullier</surname><given-names>F</given-names></name><name><surname>Gutwein</surname><given-names>M</given-names></name><name><surname>Guyer</surname><given-names>MS</given-names></name><name><surname>Habegger</surname><given-names>L</given-names></name><name><surname>Han</surname><given-names>T</given-names></name><name><surname>Henikoff</surname><given-names>JG</given-names></name><name><surname>Henz</surname><given-names>SR</given-names></name><name><surname>Hinrichs</surname><given-names>A</given-names></name><name><surname>Holster</surname><given-names>H</given-names></name><name><surname>Hyman</surname><given-names>T</given-names></name><name><surname>Iniguez</surname><given-names>AL</given-names></name><name><surname>Janette</surname><given-names>J</given-names></name><name><surname>Jensen</surname><given-names>M</given-names></name><name><surname>Kato</surname><given-names>M</given-names></name><name><surname>Kent</surname><given-names>WJ</given-names></name><name><surname>Kephart</surname><given-names>E</given-names></name><name><surname>Khivansara</surname><given-names>V</given-names></name><name><surname>Khurana</surname><given-names>E</given-names></name><name><surname>Kim</surname><given-names>JK</given-names></name><name><surname>Kolasinska-Zwierz</surname><given-names>P</given-names></name><name><surname>Lai</surname><given-names>EC</given-names></name><name><surname>Latorre</surname><given-names>I</given-names></name><name><surname>Leahey</surname><given-names>A</given-names></name><name><surname>Lewis</surname><given-names>S</given-names></name><name><surname>Lloyd</surname><given-names>P</given-names></name><name><surname>Lochovsky</surname><given-names>L</given-names></name><name><surname>Lowdon</surname><given-names>RF</given-names></name><name><surname>Lubling</surname><given-names>Y</given-names></name><name><surname>Lyne</surname><given-names>R</given-names></name><name><surname>MacCoss</surname><given-names>M</given-names></name><name><surname>Mackowiak</surname><given-names>SD</given-names></name><name><surname>Mangone</surname><given-names>M</given-names></name><name><surname>McKay</surname><given-names>S</given-names></name><name><surname>Mecenas</surname><given-names>D</given-names></name><name><surname>Merrihew</surname><given-names>G</given-names></name><name><surname>Muroyama</surname><given-names>A</given-names></name><name><surname>Murray</surname><given-names>JI</given-names></name><name><surname>Ooi</surname><given-names>SL</given-names></name><name><surname>Pham</surname><given-names>H</given-names></name><name><surname>Phippen</surname><given-names>T</given-names></name><name><surname>Preston</surname><given-names>EA</given-names></name><name><surname>Rajewsky</surname><given-names>N</given-names></name><name><surname>Rätsch</surname><given-names>G</given-names></name><name><surname>Rosenbaum</surname><given-names>H</given-names></name><name><surname>Rozowsky</surname><given-names>J</given-names></name><name><surname>Rutherford</surname><given-names>K</given-names></name><name><surname>Ruzanov</surname><given-names>P</given-names></name><name><surname>Sarov</surname><given-names>M</given-names></name><name><surname>Sasidharan</surname><given-names>R</given-names></name><name><surname>Sboner</surname><given-names>A</given-names></name><name><surname>Scheid</surname><given-names>P</given-names></name><name><surname>Segal</surname><given-names>E</given-names></name><name><surname>Shin</surname><given-names>H</given-names></name><name><surname>Shou</surname><given-names>C</given-names></name><name><surname>Slack</surname><given-names>FJ</given-names></name><name><surname>Slightam</surname><given-names>C</given-names></name><name><surname>Smith</surname><given-names>R</given-names></name><name><surname>Spencer</surname><given-names>WC</given-names></name><name><surname>Stinson</surname><given-names>EO</given-names></name><name><surname>Taing</surname><given-names>S</given-names></name><name><surname>Takasaki</surname><given-names>T</given-names></name><name><surname>Vafeados</surname><given-names>D</given-names></name><name><surname>Voronina</surname><given-names>K</given-names></name><name><surname>Wang</surname><given-names>G</given-names></name><name><surname>Washington</surname><given-names>NL</given-names></name><name><surname>Whittle</surname><given-names>CM</given-names></name><name><surname>Wu</surname><given-names>B</given-names></name><name><surname>Yan</surname><given-names>KK</given-names></name><name><surname>Zeller</surname><given-names>G</given-names></name><name><surname>Zha</surname><given-names>Z</given-names></name><name><surname>Zhong</surname><given-names>M</given-names></name><name><surname>Zhou</surname><given-names>X</given-names></name><name><surname>Ahringer</surname><given-names>J</given-names></name><name><surname>Strome</surname><given-names>S</given-names></name><name><surname>Gunsalus</surname><given-names>KC</given-names></name><name><surname>Micklem</surname><given-names>G</given-names></name><name><surname>Liu</surname><given-names>XS</given-names></name><name><surname>Reinke</surname><given-names>V</given-names></name><name><surname>Kim</surname><given-names>SK</given-names></name><name><surname>Hillier</surname><given-names>LW</given-names></name><name><surname>Henikoff</surname><given-names>S</given-names></name><name><surname>Piano</surname><given-names>F</given-names></name><name><surname>Snyder</surname><given-names>M</given-names></name><name><surname>Stein</surname><given-names>L</given-names></name><name><surname>Lieb</surname><given-names>JD</given-names></name><name><surname>Waterston</surname><given-names>RH</given-names></name></person-group><year iso-8601-date="2010">2010</year><article-title>Integrative analysis of the <italic>Caenorhabditis elegans</italic> genome by the modENCODE project</article-title><source>Science</source><volume>330</volume><fpage>1775</fpage><lpage>1787</lpage><pub-id pub-id-type="doi">10.1126/science.1196914</pub-id><pub-id pub-id-type="pmid">21177976</pub-id></element-citation></ref><ref id="bib20"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Gorkin</surname><given-names>DU</given-names></name><name><surname>Barozzi</surname><given-names>I</given-names></name><name><surname>Zhao</surname><given-names>Y</given-names></name><name><surname>Zhang</surname><given-names>Y</given-names></name><name><surname>Huang</surname><given-names>H</given-names></name><name><surname>Lee</surname><given-names>AY</given-names></name><name><surname>Li</surname><given-names>B</given-names></name><name><surname>Chiou</surname><given-names>J</given-names></name><name><surname>Wildberg</surname><given-names>A</given-names></name><name><surname>Ding</surname><given-names>B</given-names></name><name><surname>Zhang</surname><given-names>B</given-names></name><name><surname>Wang</surname><given-names>M</given-names></name><name><surname>Strattan</surname><given-names>JS</given-names></name><name><surname>Davidson</surname><given-names>JM</given-names></name><name><surname>Qiu</surname><given-names>Y</given-names></name><name><surname>Afzal</surname><given-names>V</given-names></name><name><surname>Akiyama</surname><given-names>JA</given-names></name><name><surname>Plajzer-Frick</surname><given-names>I</given-names></name><name><surname>Novak</surname><given-names>CS</given-names></name><name><surname>Kato</surname><given-names>M</given-names></name><name><surname>Garvin</surname><given-names>TH</given-names></name><name><surname>Pham</surname><given-names>QT</given-names></name><name><surname>Harrington</surname><given-names>AN</given-names></name><name><surname>Mannion</surname><given-names>BJ</given-names></name><name><surname>Lee</surname><given-names>EA</given-names></name><name><surname>Fukuda-Yuzawa</surname><given-names>Y</given-names></name><name><surname>He</surname><given-names>Y</given-names></name><name><surname>Preissl</surname><given-names>S</given-names></name><name><surname>Chee</surname><given-names>S</given-names></name><name><surname>Han</surname><given-names>JY</given-names></name><name><surname>Williams</surname><given-names>BA</given-names></name><name><surname>Trout</surname><given-names>D</given-names></name><name><surname>Amrhein</surname><given-names>H</given-names></name><name><surname>Yang</surname><given-names>H</given-names></name><name><surname>Cherry</surname><given-names>JM</given-names></name><name><surname>Wang</surname><given-names>W</given-names></name><name><surname>Gaulton</surname><given-names>K</given-names></name><name><surname>Ecker</surname><given-names>JR</given-names></name><name><surname>Shen</surname><given-names>Y</given-names></name><name><surname>Dickel</surname><given-names>DE</given-names></name><name><surname>Visel</surname><given-names>A</given-names></name><name><surname>Pennacchio</surname><given-names>LA</given-names></name><name><surname>Ren</surname><given-names>B</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>An atlas of dynamic chromatin landscapes in mouse fetal development</article-title><source>Nature</source><volume>583</volume><fpage>744</fpage><lpage>751</lpage><pub-id pub-id-type="doi">10.1038/s41586-020-2093-3</pub-id><pub-id pub-id-type="pmid">32728240</pub-id></element-citation></ref><ref id="bib21"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hnisz</surname><given-names>D</given-names></name><name><surname>Abraham</surname><given-names>BJ</given-names></name><name><surname>Lee</surname><given-names>TI</given-names></name><name><surname>Lau</surname><given-names>A</given-names></name><name><surname>Saint-André</surname><given-names>V</given-names></name><name><surname>Sigova</surname><given-names>AA</given-names></name><name><surname>Hoke</surname><given-names>HA</given-names></name><name><surname>Young</surname><given-names>RA</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>Super-enhancers in the control of cell identity and disease</article-title><source>Cell</source><volume>155</volume><fpage>934</fpage><lpage>947</lpage><pub-id pub-id-type="doi">10.1016/j.cell.2013.09.053</pub-id><pub-id pub-id-type="pmid">24119843</pub-id></element-citation></ref><ref id="bib22"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hounkpe</surname><given-names>BW</given-names></name><name><surname>Chenou</surname><given-names>F</given-names></name><name><surname>de Lima</surname><given-names>F</given-names></name><name><surname>De Paula</surname><given-names>EV</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>HRT Atlas v1.0 database: redefining human and mouse housekeeping genes and candidate reference transcripts by mining massive RNA-seq datasets</article-title><source>Nucleic Acids Research</source><volume>49</volume><fpage>D947</fpage><lpage>D955</lpage><pub-id pub-id-type="doi">10.1093/nar/gkaa609</pub-id><pub-id pub-id-type="pmid">32663312</pub-id></element-citation></ref><ref id="bib23"><element-citation publication-type="software"><person-group person-group-type="author"><name><surname>Hudaiberdiev</surname><given-names>S</given-names></name></person-group><year iso-8601-date="2024">2024</year><data-title>HOT</data-title><version designator="swh:1:rev:9510b67053054a4cb97ea747290ad3e913e180f5">swh:1:rev:9510b67053054a4cb97ea747290ad3e913e180f5</version><source>Software Heritage</source><ext-link ext-link-type="uri" xlink:href="https://archive.softwareheritage.org/swh:1:dir:d3a0344f53442a06060b03b8a37941bba5391078;origin=https://github.com/okurman/HOT;visit=swh:1:snp:050692d71432c06a19a094a02439b8d5bcc2a394;anchor=swh:1:rev:9510b67053054a4cb97ea747290ad3e913e180f5">https://archive.softwareheritage.org/swh:1:dir:d3a0344f53442a06060b03b8a37941bba5391078;origin=https://github.com/okurman/HOT;visit=swh:1:snp:050692d71432c06a19a094a02439b8d5bcc2a394;anchor=swh:1:rev:9510b67053054a4cb97ea747290ad3e913e180f5</ext-link></element-citation></ref><ref id="bib24"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Hunter</surname><given-names>JD</given-names></name></person-group><year iso-8601-date="2007">2007</year><article-title>Matplotlib: a 2d graphics environment</article-title><source>Computing in Science &amp; Engineering</source><volume>9</volume><fpage>90</fpage><lpage>95</lpage><pub-id pub-id-type="doi">10.1109/MCSE.2007.55</pub-id></element-citation></ref><ref id="bib25"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Karczewski</surname><given-names>KJ</given-names></name><name><surname>Francioli</surname><given-names>LC</given-names></name><name><surname>Tiao</surname><given-names>G</given-names></name><name><surname>Cummings</surname><given-names>BB</given-names></name><name><surname>Alföldi</surname><given-names>J</given-names></name><name><surname>Wang</surname><given-names>Q</given-names></name><name><surname>Collins</surname><given-names>RL</given-names></name><name><surname>Laricchia</surname><given-names>KM</given-names></name><name><surname>Ganna</surname><given-names>A</given-names></name><name><surname>Birnbaum</surname><given-names>DP</given-names></name><name><surname>Gauthier</surname><given-names>LD</given-names></name><name><surname>Brand</surname><given-names>H</given-names></name><name><surname>Solomonson</surname><given-names>M</given-names></name><name><surname>Watts</surname><given-names>NA</given-names></name><name><surname>Rhodes</surname><given-names>D</given-names></name><name><surname>Singer-Berk</surname><given-names>M</given-names></name><name><surname>England</surname><given-names>EM</given-names></name><name><surname>Seaby</surname><given-names>EG</given-names></name><name><surname>Kosmicki</surname><given-names>JA</given-names></name><name><surname>Walters</surname><given-names>RK</given-names></name><name><surname>Tashman</surname><given-names>K</given-names></name><name><surname>Farjoun</surname><given-names>Y</given-names></name><name><surname>Banks</surname><given-names>E</given-names></name><name><surname>Poterba</surname><given-names>T</given-names></name><name><surname>Wang</surname><given-names>A</given-names></name><name><surname>Seed</surname><given-names>C</given-names></name><name><surname>Whiffin</surname><given-names>N</given-names></name><name><surname>Chong</surname><given-names>JX</given-names></name><name><surname>Samocha</surname><given-names>KE</given-names></name><name><surname>Pierce-Hoffman</surname><given-names>E</given-names></name><name><surname>Zappala</surname><given-names>Z</given-names></name><name><surname>O’Donnell-Luria</surname><given-names>AH</given-names></name><name><surname>Minikel</surname><given-names>EV</given-names></name><name><surname>Weisburd</surname><given-names>B</given-names></name><name><surname>Lek</surname><given-names>M</given-names></name><name><surname>Ware</surname><given-names>JS</given-names></name><name><surname>Vittal</surname><given-names>C</given-names></name><name><surname>Armean</surname><given-names>IM</given-names></name><name><surname>Bergelson</surname><given-names>L</given-names></name><name><surname>Cibulskis</surname><given-names>K</given-names></name><name><surname>Connolly</surname><given-names>KM</given-names></name><name><surname>Covarrubias</surname><given-names>M</given-names></name><name><surname>Donnelly</surname><given-names>S</given-names></name><name><surname>Ferriera</surname><given-names>S</given-names></name><name><surname>Gabriel</surname><given-names>S</given-names></name><name><surname>Gentry</surname><given-names>J</given-names></name><name><surname>Gupta</surname><given-names>N</given-names></name><name><surname>Jeandet</surname><given-names>T</given-names></name><name><surname>Kaplan</surname><given-names>D</given-names></name><name><surname>Llanwarne</surname><given-names>C</given-names></name><name><surname>Munshi</surname><given-names>R</given-names></name><name><surname>Novod</surname><given-names>S</given-names></name><name><surname>Petrillo</surname><given-names>N</given-names></name><name><surname>Roazen</surname><given-names>D</given-names></name><name><surname>Ruano-Rubio</surname><given-names>V</given-names></name><name><surname>Saltzman</surname><given-names>A</given-names></name><name><surname>Schleicher</surname><given-names>M</given-names></name><name><surname>Soto</surname><given-names>J</given-names></name><name><surname>Tibbetts</surname><given-names>K</given-names></name><name><surname>Tolonen</surname><given-names>C</given-names></name><name><surname>Wade</surname><given-names>G</given-names></name><name><surname>Talkowski</surname><given-names>ME</given-names></name><name><surname>Neale</surname><given-names>BM</given-names></name><name><surname>Daly</surname><given-names>MJ</given-names></name><name><surname>MacArthur</surname><given-names>DG</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>The mutational constraint spectrum quantified from variation in 141,456 humans</article-title><source>Nature</source><volume>581</volume><fpage>434</fpage><lpage>443</lpage><pub-id pub-id-type="doi">10.1038/s41586-020-2308-7</pub-id><pub-id pub-id-type="pmid">32461654</pub-id></element-citation></ref><ref id="bib26"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Kvon</surname><given-names>EZ</given-names></name><name><surname>Stampfel</surname><given-names>G</given-names></name><name><surname>Yáñez-Cuna</surname><given-names>JO</given-names></name><name><surname>Dickson</surname><given-names>BJ</given-names></name><name><surname>Stark</surname><given-names>A</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>HOT regions function as patterned developmental enhancers and have a distinct cis-regulatory signature</article-title><source>Genes &amp; Development</source><volume>26</volume><fpage>908</fpage><lpage>913</lpage><pub-id pub-id-type="doi">10.1101/gad.188052.112</pub-id><pub-id pub-id-type="pmid">22499593</pub-id></element-citation></ref><ref id="bib27"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Lambert</surname><given-names>SA</given-names></name><name><surname>Jolma</surname><given-names>A</given-names></name><name><surname>Campitelli</surname><given-names>LF</given-names></name><name><surname>Das</surname><given-names>PK</given-names></name><name><surname>Yin</surname><given-names>Y</given-names></name><name><surname>Albu</surname><given-names>M</given-names></name><name><surname>Chen</surname><given-names>X</given-names></name><name><surname>Taipale</surname><given-names>J</given-names></name><name><surname>Hughes</surname><given-names>TR</given-names></name><name><surname>Weirauch</surname><given-names>MT</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>The human transcription factors</article-title><source>Cell</source><volume>172</volume><fpage>650</fpage><lpage>665</lpage><pub-id pub-id-type="doi">10.1016/j.cell.2018.01.029</pub-id><pub-id pub-id-type="pmid">29425488</pub-id></element-citation></ref><ref id="bib28"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Lee</surname><given-names>D</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>LS-GKM: a new GKM-SVM for large-scale datasets</article-title><source>Bioinformatics</source><volume>32</volume><fpage>2196</fpage><lpage>2198</lpage><pub-id pub-id-type="doi">10.1093/bioinformatics/btw142</pub-id><pub-id pub-id-type="pmid">27153584</pub-id></element-citation></ref><ref id="bib29"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Lee</surname><given-names>R</given-names></name><name><surname>Kang</surname><given-names>MK</given-names></name><name><surname>Kim</surname><given-names>YJ</given-names></name><name><surname>Yang</surname><given-names>B</given-names></name><name><surname>Shim</surname><given-names>H</given-names></name><name><surname>Kim</surname><given-names>S</given-names></name><name><surname>Kim</surname><given-names>K</given-names></name><name><surname>Yang</surname><given-names>CM</given-names></name><name><surname>Min</surname><given-names>BG</given-names></name><name><surname>Jung</surname><given-names>WJ</given-names></name><name><surname>Lee</surname><given-names>EC</given-names></name><name><surname>Joo</surname><given-names>JS</given-names></name><name><surname>Park</surname><given-names>G</given-names></name><name><surname>Cho</surname><given-names>WK</given-names></name><name><surname>Kim</surname><given-names>HP</given-names></name></person-group><year iso-8601-date="2022">2022</year><article-title>CTCF-mediated chromatin looping provides a topological framework for the formation of phase-separated transcriptional condensates</article-title><source>Nucleic Acids Research</source><volume>50</volume><fpage>207</fpage><lpage>226</lpage><pub-id pub-id-type="doi">10.1093/nar/gkab1242</pub-id><pub-id pub-id-type="pmid">34931241</pub-id></element-citation></ref><ref id="bib30"><element-citation publication-type="software"><person-group person-group-type="author"><name><surname>Lee</surname><given-names>D</given-names></name></person-group><year iso-8601-date="2023">2023</year><data-title>Lsgkm</data-title><version designator="3d92f3f">3d92f3f</version><source>GitHub</source><ext-link ext-link-type="uri" xlink:href="https://github.com/Dongwon-Lee/lsgkm">https://github.com/Dongwon-Lee/lsgkm</ext-link></element-citation></ref><ref id="bib31"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Li</surname><given-names>H</given-names></name></person-group><year iso-8601-date="2011">2011</year><article-title>A statistical framework for SNP calling, mutation discovery, association mapping and population genetical parameter estimation from sequencing data</article-title><source>Bioinformatics</source><volume>27</volume><fpage>2987</fpage><lpage>2993</lpage><pub-id pub-id-type="doi">10.1093/bioinformatics/btr509</pub-id><pub-id pub-id-type="pmid">21903627</pub-id></element-citation></ref><ref id="bib32"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Lieberman-Aiden</surname><given-names>E</given-names></name><name><surname>van Berkum</surname><given-names>NL</given-names></name><name><surname>Williams</surname><given-names>L</given-names></name><name><surname>Imakaev</surname><given-names>M</given-names></name><name><surname>Ragoczy</surname><given-names>T</given-names></name><name><surname>Telling</surname><given-names>A</given-names></name><name><surname>Amit</surname><given-names>I</given-names></name><name><surname>Lajoie</surname><given-names>BR</given-names></name><name><surname>Sabo</surname><given-names>PJ</given-names></name><name><surname>Dorschner</surname><given-names>MO</given-names></name><name><surname>Sandstrom</surname><given-names>R</given-names></name><name><surname>Bernstein</surname><given-names>B</given-names></name><name><surname>Bender</surname><given-names>MA</given-names></name><name><surname>Groudine</surname><given-names>M</given-names></name><name><surname>Gnirke</surname><given-names>A</given-names></name><name><surname>Stamatoyannopoulos</surname><given-names>J</given-names></name><name><surname>Mirny</surname><given-names>LA</given-names></name><name><surname>Lander</surname><given-names>ES</given-names></name><name><surname>Dekker</surname><given-names>J</given-names></name></person-group><year iso-8601-date="2009">2009</year><article-title>Comprehensive mapping of long-range interactions reveals folding principles of the human genome</article-title><source>Science</source><volume>326</volume><fpage>289</fpage><lpage>293</lpage><pub-id pub-id-type="doi">10.1126/science.1181369</pub-id><pub-id pub-id-type="pmid">19815776</pub-id></element-citation></ref><ref id="bib33"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Liu</surname><given-names>J</given-names></name><name><surname>Miao</surname><given-names>X</given-names></name><name><surname>Xiao</surname><given-names>B</given-names></name><name><surname>Huang</surname><given-names>J</given-names></name><name><surname>Tao</surname><given-names>X</given-names></name><name><surname>Zhang</surname><given-names>J</given-names></name><name><surname>Zhao</surname><given-names>H</given-names></name><name><surname>Pan</surname><given-names>Y</given-names></name><name><surname>Wang</surname><given-names>H</given-names></name><name><surname>Gao</surname><given-names>G</given-names></name><name><surname>Xiao</surname><given-names>GG</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>Obg-like atpase 1 enhances chemoresistance of breast cancer <italic>via</italic> activation of tgf-β/smad axis cascades</article-title><source>Frontiers in Pharmacology</source><volume>11</volume><elocation-id>666</elocation-id><pub-id pub-id-type="doi">10.3389/fphar.2020.00666</pub-id><pub-id pub-id-type="pmid">32528278</pub-id></element-citation></ref><ref id="bib34"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Lizio</surname><given-names>M</given-names></name><name><surname>Abugessaisa</surname><given-names>I</given-names></name><name><surname>Noguchi</surname><given-names>S</given-names></name><name><surname>Kondo</surname><given-names>A</given-names></name><name><surname>Hasegawa</surname><given-names>A</given-names></name><name><surname>Hon</surname><given-names>CC</given-names></name><name><surname>de Hoon</surname><given-names>M</given-names></name><name><surname>Severin</surname><given-names>J</given-names></name><name><surname>Oki</surname><given-names>S</given-names></name><name><surname>Hayashizaki</surname><given-names>Y</given-names></name><name><surname>Carninci</surname><given-names>P</given-names></name><name><surname>Kasukawa</surname><given-names>T</given-names></name><name><surname>Kawaji</surname><given-names>H</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Update of the FANTOM web resource: expansion to provide additional transcriptome atlases</article-title><source>Nucleic Acids Research</source><volume>47</volume><fpage>D752</fpage><lpage>D758</lpage><pub-id pub-id-type="doi">10.1093/nar/gky1099</pub-id><pub-id pub-id-type="pmid">30407557</pub-id></element-citation></ref><ref id="bib35"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Long</surname><given-names>HK</given-names></name><name><surname>Prescott</surname><given-names>SL</given-names></name><name><surname>Wysocka</surname><given-names>J</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>Ever-changing landscapes: transcriptional enhancers in development and evolution</article-title><source>Cell</source><volume>167</volume><fpage>1170</fpage><lpage>1187</lpage><pub-id pub-id-type="doi">10.1016/j.cell.2016.09.018</pub-id><pub-id pub-id-type="pmid">27863239</pub-id></element-citation></ref><ref id="bib36"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Merika</surname><given-names>M</given-names></name><name><surname>Thanos</surname><given-names>D</given-names></name></person-group><year iso-8601-date="2001">2001</year><article-title>Enhanceosomes</article-title><source>Current Opinion in Genetics &amp; Development</source><volume>11</volume><fpage>205</fpage><lpage>208</lpage><pub-id pub-id-type="doi">10.1016/s0959-437x(00)00180-5</pub-id><pub-id pub-id-type="pmid">11250145</pub-id></element-citation></ref><ref id="bib37"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Michailidou</surname><given-names>K</given-names></name><name><surname>Lindström</surname><given-names>S</given-names></name><name><surname>Dennis</surname><given-names>J</given-names></name><name><surname>Beesley</surname><given-names>J</given-names></name><name><surname>Hui</surname><given-names>S</given-names></name><name><surname>Kar</surname><given-names>S</given-names></name><name><surname>Lemaçon</surname><given-names>A</given-names></name><name><surname>Soucy</surname><given-names>P</given-names></name><name><surname>Glubb</surname><given-names>D</given-names></name><name><surname>Rostamianfar</surname><given-names>A</given-names></name><name><surname>Bolla</surname><given-names>MK</given-names></name><name><surname>Wang</surname><given-names>Q</given-names></name><name><surname>Tyrer</surname><given-names>J</given-names></name><name><surname>Dicks</surname><given-names>E</given-names></name><name><surname>Lee</surname><given-names>A</given-names></name><name><surname>Wang</surname><given-names>Z</given-names></name><name><surname>Allen</surname><given-names>J</given-names></name><name><surname>Keeman</surname><given-names>R</given-names></name><name><surname>Eilber</surname><given-names>U</given-names></name><name><surname>French</surname><given-names>JD</given-names></name><name><surname>Qing Chen</surname><given-names>X</given-names></name><name><surname>Fachal</surname><given-names>L</given-names></name><name><surname>McCue</surname><given-names>K</given-names></name><name><surname>McCart Reed</surname><given-names>AE</given-names></name><name><surname>Ghoussaini</surname><given-names>M</given-names></name><name><surname>Carroll</surname><given-names>JS</given-names></name><name><surname>Jiang</surname><given-names>X</given-names></name><name><surname>Finucane</surname><given-names>H</given-names></name><name><surname>Adams</surname><given-names>M</given-names></name><name><surname>Adank</surname><given-names>MA</given-names></name><name><surname>Ahsan</surname><given-names>H</given-names></name><name><surname>Aittomäki</surname><given-names>K</given-names></name><name><surname>Anton-Culver</surname><given-names>H</given-names></name><name><surname>Antonenkova</surname><given-names>NN</given-names></name><name><surname>Arndt</surname><given-names>V</given-names></name><name><surname>Aronson</surname><given-names>KJ</given-names></name><name><surname>Arun</surname><given-names>B</given-names></name><name><surname>Auer</surname><given-names>PL</given-names></name><name><surname>Bacot</surname><given-names>F</given-names></name><name><surname>Barrdahl</surname><given-names>M</given-names></name><name><surname>Baynes</surname><given-names>C</given-names></name><name><surname>Beckmann</surname><given-names>MW</given-names></name><name><surname>Behrens</surname><given-names>S</given-names></name><name><surname>Benitez</surname><given-names>J</given-names></name><name><surname>Bermisheva</surname><given-names>M</given-names></name><name><surname>Bernstein</surname><given-names>L</given-names></name><name><surname>Blomqvist</surname><given-names>C</given-names></name><name><surname>Bogdanova</surname><given-names>NV</given-names></name><name><surname>Bojesen</surname><given-names>SE</given-names></name><name><surname>Bonanni</surname><given-names>B</given-names></name><name><surname>Børresen-Dale</surname><given-names>A-L</given-names></name><name><surname>Brand</surname><given-names>JS</given-names></name><name><surname>Brauch</surname><given-names>H</given-names></name><name><surname>Brennan</surname><given-names>P</given-names></name><name><surname>Brenner</surname><given-names>H</given-names></name><name><surname>Brinton</surname><given-names>L</given-names></name><name><surname>Broberg</surname><given-names>P</given-names></name><name><surname>Brock</surname><given-names>IW</given-names></name><name><surname>Broeks</surname><given-names>A</given-names></name><name><surname>Brooks-Wilson</surname><given-names>A</given-names></name><name><surname>Brucker</surname><given-names>SY</given-names></name><name><surname>Brüning</surname><given-names>T</given-names></name><name><surname>Burwinkel</surname><given-names>B</given-names></name><name><surname>Butterbach</surname><given-names>K</given-names></name><name><surname>Cai</surname><given-names>Q</given-names></name><name><surname>Cai</surname><given-names>H</given-names></name><name><surname>Caldés</surname><given-names>T</given-names></name><name><surname>Canzian</surname><given-names>F</given-names></name><name><surname>Carracedo</surname><given-names>A</given-names></name><name><surname>Carter</surname><given-names>BD</given-names></name><name><surname>Castelao</surname><given-names>JE</given-names></name><name><surname>Chan</surname><given-names>TL</given-names></name><name><surname>David Cheng</surname><given-names>T-Y</given-names></name><name><surname>Seng Chia</surname><given-names>K</given-names></name><name><surname>Choi</surname><given-names>J-Y</given-names></name><name><surname>Christiansen</surname><given-names>H</given-names></name><name><surname>Clarke</surname><given-names>CL</given-names></name><collab>NBCS Collaborators</collab><name><surname>Collée</surname><given-names>M</given-names></name><name><surname>Conroy</surname><given-names>DM</given-names></name><name><surname>Cordina-Duverger</surname><given-names>E</given-names></name><name><surname>Cornelissen</surname><given-names>S</given-names></name><name><surname>Cox</surname><given-names>DG</given-names></name><name><surname>Cox</surname><given-names>A</given-names></name><name><surname>Cross</surname><given-names>SS</given-names></name><name><surname>Cunningham</surname><given-names>JM</given-names></name><name><surname>Czene</surname><given-names>K</given-names></name><name><surname>Daly</surname><given-names>MB</given-names></name><name><surname>Devilee</surname><given-names>P</given-names></name><name><surname>Doheny</surname><given-names>KF</given-names></name><name><surname>Dörk</surname><given-names>T</given-names></name><name><surname>Dos-Santos-Silva</surname><given-names>I</given-names></name><name><surname>Dumont</surname><given-names>M</given-names></name><name><surname>Durcan</surname><given-names>L</given-names></name><name><surname>Dwek</surname><given-names>M</given-names></name><name><surname>Eccles</surname><given-names>DM</given-names></name><name><surname>Ekici</surname><given-names>AB</given-names></name><name><surname>Eliassen</surname><given-names>AH</given-names></name><name><surname>Ellberg</surname><given-names>C</given-names></name><name><surname>Elvira</surname><given-names>M</given-names></name><name><surname>Engel</surname><given-names>C</given-names></name><name><surname>Eriksson</surname><given-names>M</given-names></name><name><surname>Fasching</surname><given-names>PA</given-names></name><name><surname>Figueroa</surname><given-names>J</given-names></name><name><surname>Flesch-Janys</surname><given-names>D</given-names></name><name><surname>Fletcher</surname><given-names>O</given-names></name><name><surname>Flyger</surname><given-names>H</given-names></name><name><surname>Fritschi</surname><given-names>L</given-names></name><name><surname>Gaborieau</surname><given-names>V</given-names></name><name><surname>Gabrielson</surname><given-names>M</given-names></name><name><surname>Gago-Dominguez</surname><given-names>M</given-names></name><name><surname>Gao</surname><given-names>Y-T</given-names></name><name><surname>Gapstur</surname><given-names>SM</given-names></name><name><surname>García-Sáenz</surname><given-names>JA</given-names></name><name><surname>Gaudet</surname><given-names>MM</given-names></name><name><surname>Georgoulias</surname><given-names>V</given-names></name><name><surname>Giles</surname><given-names>GG</given-names></name><name><surname>Glendon</surname><given-names>G</given-names></name><name><surname>Goldberg</surname><given-names>MS</given-names></name><name><surname>Goldgar</surname><given-names>DE</given-names></name><name><surname>González-Neira</surname><given-names>A</given-names></name><name><surname>Grenaker Alnæs</surname><given-names>GI</given-names></name><name><surname>Grip</surname><given-names>M</given-names></name><name><surname>Gronwald</surname><given-names>J</given-names></name><name><surname>Grundy</surname><given-names>A</given-names></name><name><surname>Guénel</surname><given-names>P</given-names></name><name><surname>Haeberle</surname><given-names>L</given-names></name><name><surname>Hahnen</surname><given-names>E</given-names></name><name><surname>Haiman</surname><given-names>CA</given-names></name><name><surname>Håkansson</surname><given-names>N</given-names></name><name><surname>Hamann</surname><given-names>U</given-names></name><name><surname>Hamel</surname><given-names>N</given-names></name><name><surname>Hankinson</surname><given-names>S</given-names></name><name><surname>Harrington</surname><given-names>P</given-names></name><name><surname>Hart</surname><given-names>SN</given-names></name><name><surname>Hartikainen</surname><given-names>JM</given-names></name><name><surname>Hartman</surname><given-names>M</given-names></name><name><surname>Hein</surname><given-names>A</given-names></name><name><surname>Heyworth</surname><given-names>J</given-names></name><name><surname>Hicks</surname><given-names>B</given-names></name><name><surname>Hillemanns</surname><given-names>P</given-names></name><name><surname>Ho</surname><given-names>DN</given-names></name><name><surname>Hollestelle</surname><given-names>A</given-names></name><name><surname>Hooning</surname><given-names>MJ</given-names></name><name><surname>Hoover</surname><given-names>RN</given-names></name><name><surname>Hopper</surname><given-names>JL</given-names></name><name><surname>Hou</surname><given-names>M-F</given-names></name><name><surname>Hsiung</surname><given-names>C-N</given-names></name><name><surname>Huang</surname><given-names>G</given-names></name><name><surname>Humphreys</surname><given-names>K</given-names></name><name><surname>Ishiguro</surname><given-names>J</given-names></name><name><surname>Ito</surname><given-names>H</given-names></name><name><surname>Iwasaki</surname><given-names>M</given-names></name><name><surname>Iwata</surname><given-names>H</given-names></name><name><surname>Jakubowska</surname><given-names>A</given-names></name><name><surname>Janni</surname><given-names>W</given-names></name><name><surname>John</surname><given-names>EM</given-names></name><name><surname>Johnson</surname><given-names>N</given-names></name><name><surname>Jones</surname><given-names>K</given-names></name><name><surname>Jones</surname><given-names>M</given-names></name><name><surname>Jukkola-Vuorinen</surname><given-names>A</given-names></name><name><surname>Kaaks</surname><given-names>R</given-names></name><name><surname>Kabisch</surname><given-names>M</given-names></name><name><surname>Kaczmarek</surname><given-names>K</given-names></name><name><surname>Kang</surname><given-names>D</given-names></name><name><surname>Kasuga</surname><given-names>Y</given-names></name><name><surname>Kerin</surname><given-names>MJ</given-names></name><name><surname>Khan</surname><given-names>S</given-names></name><name><surname>Khusnutdinova</surname><given-names>E</given-names></name><name><surname>Kiiski</surname><given-names>JI</given-names></name><name><surname>Kim</surname><given-names>S-W</given-names></name><name><surname>Knight</surname><given-names>JA</given-names></name><name><surname>Kosma</surname><given-names>V-M</given-names></name><name><surname>Kristensen</surname><given-names>VN</given-names></name><name><surname>Krüger</surname><given-names>U</given-names></name><name><surname>Kwong</surname><given-names>A</given-names></name><name><surname>Lambrechts</surname><given-names>D</given-names></name><name><surname>Le Marchand</surname><given-names>L</given-names></name><name><surname>Lee</surname><given-names>E</given-names></name><name><surname>Lee</surname><given-names>MH</given-names></name><name><surname>Lee</surname><given-names>JW</given-names></name><name><surname>Neng Lee</surname><given-names>C</given-names></name><name><surname>Lejbkowicz</surname><given-names>F</given-names></name><name><surname>Li</surname><given-names>J</given-names></name><name><surname>Lilyquist</surname><given-names>J</given-names></name><name><surname>Lindblom</surname><given-names>A</given-names></name><name><surname>Lissowska</surname><given-names>J</given-names></name><name><surname>Lo</surname><given-names>W-Y</given-names></name><name><surname>Loibl</surname><given-names>S</given-names></name><name><surname>Long</surname><given-names>J</given-names></name><name><surname>Lophatananon</surname><given-names>A</given-names></name><name><surname>Lubinski</surname><given-names>J</given-names></name><name><surname>Luccarini</surname><given-names>C</given-names></name><name><surname>Lux</surname><given-names>MP</given-names></name><name><surname>Ma</surname><given-names>ESK</given-names></name><name><surname>MacInnis</surname><given-names>RJ</given-names></name><name><surname>Maishman</surname><given-names>T</given-names></name><name><surname>Makalic</surname><given-names>E</given-names></name><name><surname>Malone</surname><given-names>KE</given-names></name><name><surname>Kostovska</surname><given-names>IM</given-names></name><name><surname>Mannermaa</surname><given-names>A</given-names></name><name><surname>Manoukian</surname><given-names>S</given-names></name><name><surname>Manson</surname><given-names>JE</given-names></name><name><surname>Margolin</surname><given-names>S</given-names></name><name><surname>Mariapun</surname><given-names>S</given-names></name><name><surname>Martinez</surname><given-names>ME</given-names></name><name><surname>Matsuo</surname><given-names>K</given-names></name><name><surname>Mavroudis</surname><given-names>D</given-names></name><name><surname>McKay</surname><given-names>J</given-names></name><name><surname>McLean</surname><given-names>C</given-names></name><name><surname>Meijers-Heijboer</surname><given-names>H</given-names></name><name><surname>Meindl</surname><given-names>A</given-names></name><name><surname>Menéndez</surname><given-names>P</given-names></name><name><surname>Menon</surname><given-names>U</given-names></name><name><surname>Meyer</surname><given-names>J</given-names></name><name><surname>Miao</surname><given-names>H</given-names></name><name><surname>Miller</surname><given-names>N</given-names></name><name><surname>Taib</surname><given-names>NAM</given-names></name><name><surname>Muir</surname><given-names>K</given-names></name><name><surname>Mulligan</surname><given-names>AM</given-names></name><name><surname>Mulot</surname><given-names>C</given-names></name><name><surname>Neuhausen</surname><given-names>SL</given-names></name><name><surname>Nevanlinna</surname><given-names>H</given-names></name><name><surname>Neven</surname><given-names>P</given-names></name><name><surname>Nielsen</surname><given-names>SF</given-names></name><name><surname>Noh</surname><given-names>D-Y</given-names></name><name><surname>Nordestgaard</surname><given-names>BG</given-names></name><name><surname>Norman</surname><given-names>A</given-names></name><name><surname>Olopade</surname><given-names>OI</given-names></name><name><surname>Olson</surname><given-names>JE</given-names></name><name><surname>Olsson</surname><given-names>H</given-names></name><name><surname>Olswold</surname><given-names>C</given-names></name><name><surname>Orr</surname><given-names>N</given-names></name><name><surname>Pankratz</surname><given-names>VS</given-names></name><name><surname>Park</surname><given-names>SK</given-names></name><name><surname>Park-Simon</surname><given-names>T-W</given-names></name><name><surname>Lloyd</surname><given-names>R</given-names></name><name><surname>Perez</surname><given-names>JIA</given-names></name><name><surname>Peterlongo</surname><given-names>P</given-names></name><name><surname>Peto</surname><given-names>J</given-names></name><name><surname>Phillips</surname><given-names>K-A</given-names></name><name><surname>Pinchev</surname><given-names>M</given-names></name><name><surname>Plaseska-Karanfilska</surname><given-names>D</given-names></name><name><surname>Prentice</surname><given-names>R</given-names></name><name><surname>Presneau</surname><given-names>N</given-names></name><name><surname>Prokofyeva</surname><given-names>D</given-names></name><name><surname>Pugh</surname><given-names>E</given-names></name><name><surname>Pylkäs</surname><given-names>K</given-names></name><name><surname>Rack</surname><given-names>B</given-names></name><name><surname>Radice</surname><given-names>P</given-names></name><name><surname>Rahman</surname><given-names>N</given-names></name><name><surname>Rennert</surname><given-names>G</given-names></name><name><surname>Rennert</surname><given-names>HS</given-names></name><name><surname>Rhenius</surname><given-names>V</given-names></name><name><surname>Romero</surname><given-names>A</given-names></name><name><surname>Romm</surname><given-names>J</given-names></name><name><surname>Ruddy</surname><given-names>KJ</given-names></name><name><surname>Rüdiger</surname><given-names>T</given-names></name><name><surname>Rudolph</surname><given-names>A</given-names></name><name><surname>Ruebner</surname><given-names>M</given-names></name><name><surname>Rutgers</surname><given-names>EJT</given-names></name><name><surname>Saloustros</surname><given-names>E</given-names></name><name><surname>Sandler</surname><given-names>DP</given-names></name><name><surname>Sangrajrang</surname><given-names>S</given-names></name><name><surname>Sawyer</surname><given-names>EJ</given-names></name><name><surname>Schmidt</surname><given-names>DF</given-names></name><name><surname>Schmutzler</surname><given-names>RK</given-names></name><name><surname>Schneeweiss</surname><given-names>A</given-names></name><name><surname>Schoemaker</surname><given-names>MJ</given-names></name><name><surname>Schumacher</surname><given-names>F</given-names></name><name><surname>Schürmann</surname><given-names>P</given-names></name><name><surname>Scott</surname><given-names>RJ</given-names></name><name><surname>Scott</surname><given-names>C</given-names></name><name><surname>Seal</surname><given-names>S</given-names></name><name><surname>Seynaeve</surname><given-names>C</given-names></name><name><surname>Shah</surname><given-names>M</given-names></name><name><surname>Sharma</surname><given-names>P</given-names></name><name><surname>Shen</surname><given-names>C-Y</given-names></name><name><surname>Sheng</surname><given-names>G</given-names></name><name><surname>Sherman</surname><given-names>ME</given-names></name><name><surname>Shrubsole</surname><given-names>MJ</given-names></name><name><surname>Shu</surname><given-names>X-O</given-names></name><name><surname>Smeets</surname><given-names>A</given-names></name><name><surname>Sohn</surname><given-names>C</given-names></name><name><surname>Southey</surname><given-names>MC</given-names></name><name><surname>Spinelli</surname><given-names>JJ</given-names></name><name><surname>Stegmaier</surname><given-names>C</given-names></name><name><surname>Stewart-Brown</surname><given-names>S</given-names></name><name><surname>Stone</surname><given-names>J</given-names></name><name><surname>Stram</surname><given-names>DO</given-names></name><name><surname>Surowy</surname><given-names>H</given-names></name><name><surname>Swerdlow</surname><given-names>A</given-names></name><name><surname>Tamimi</surname><given-names>R</given-names></name><name><surname>Taylor</surname><given-names>JA</given-names></name><name><surname>Tengström</surname><given-names>M</given-names></name><name><surname>Teo</surname><given-names>SH</given-names></name><name><surname>Beth Terry</surname><given-names>M</given-names></name><name><surname>Tessier</surname><given-names>DC</given-names></name><name><surname>Thanasitthichai</surname><given-names>S</given-names></name><name><surname>Thöne</surname><given-names>K</given-names></name><name><surname>Tollenaar</surname><given-names>RAEM</given-names></name><name><surname>Tomlinson</surname><given-names>I</given-names></name><name><surname>Tong</surname><given-names>L</given-names></name><name><surname>Torres</surname><given-names>D</given-names></name><name><surname>Truong</surname><given-names>T</given-names></name><name><surname>Tseng</surname><given-names>C-C</given-names></name><name><surname>Tsugane</surname><given-names>S</given-names></name><name><surname>Ulmer</surname><given-names>H-U</given-names></name><name><surname>Ursin</surname><given-names>G</given-names></name><name><surname>Untch</surname><given-names>M</given-names></name><name><surname>Vachon</surname><given-names>C</given-names></name><name><surname>van Asperen</surname><given-names>CJ</given-names></name><name><surname>Van Den Berg</surname><given-names>D</given-names></name><name><surname>van den Ouweland</surname><given-names>AMW</given-names></name><name><surname>van der Kolk</surname><given-names>L</given-names></name><name><surname>van der Luijt</surname><given-names>RB</given-names></name><name><surname>Vincent</surname><given-names>D</given-names></name><name><surname>Vollenweider</surname><given-names>J</given-names></name><name><surname>Waisfisz</surname><given-names>Q</given-names></name><name><surname>Wang-Gohrke</surname><given-names>S</given-names></name><name><surname>Weinberg</surname><given-names>CR</given-names></name><name><surname>Wendt</surname><given-names>C</given-names></name><name><surname>Whittemore</surname><given-names>AS</given-names></name><name><surname>Wildiers</surname><given-names>H</given-names></name><name><surname>Willett</surname><given-names>W</given-names></name><name><surname>Winqvist</surname><given-names>R</given-names></name><name><surname>Wolk</surname><given-names>A</given-names></name><name><surname>Wu</surname><given-names>AH</given-names></name><name><surname>Xia</surname><given-names>L</given-names></name><name><surname>Yamaji</surname><given-names>T</given-names></name><name><surname>Yang</surname><given-names>XR</given-names></name><name><surname>Har Yip</surname><given-names>C</given-names></name><name><surname>Yoo</surname><given-names>K-Y</given-names></name><name><surname>Yu</surname><given-names>J-C</given-names></name><name><surname>Zheng</surname><given-names>W</given-names></name><name><surname>Zheng</surname><given-names>Y</given-names></name><name><surname>Zhu</surname><given-names>B</given-names></name><name><surname>Ziogas</surname><given-names>A</given-names></name><name><surname>Ziv</surname><given-names>E</given-names></name><collab>ABCTB Investigators</collab><collab>ConFab/AOCS Investigators</collab><name><surname>Lakhani</surname><given-names>SR</given-names></name><name><surname>Antoniou</surname><given-names>AC</given-names></name><name><surname>Droit</surname><given-names>A</given-names></name><name><surname>Andrulis</surname><given-names>IL</given-names></name><name><surname>Amos</surname><given-names>CI</given-names></name><name><surname>Couch</surname><given-names>FJ</given-names></name><name><surname>Pharoah</surname><given-names>PDP</given-names></name><name><surname>Chang-Claude</surname><given-names>J</given-names></name><name><surname>Hall</surname><given-names>P</given-names></name><name><surname>Hunter</surname><given-names>DJ</given-names></name><name><surname>Milne</surname><given-names>RL</given-names></name><name><surname>García-Closas</surname><given-names>M</given-names></name><name><surname>Schmidt</surname><given-names>MK</given-names></name><name><surname>Chanock</surname><given-names>SJ</given-names></name><name><surname>Dunning</surname><given-names>AM</given-names></name><name><surname>Edwards</surname><given-names>SL</given-names></name><name><surname>Bader</surname><given-names>GD</given-names></name><name><surname>Chenevix-Trench</surname><given-names>G</given-names></name><name><surname>Simard</surname><given-names>J</given-names></name><name><surname>Kraft</surname><given-names>P</given-names></name><name><surname>Easton</surname><given-names>DF</given-names></name></person-group><year iso-8601-date="2017">2017</year><article-title>Association analysis identifies 65 new breast cancer risk loci</article-title><source>Nature</source><volume>551</volume><fpage>92</fpage><lpage>94</lpage><pub-id pub-id-type="doi">10.1038/nature24284</pub-id><pub-id pub-id-type="pmid">29059683</pub-id></element-citation></ref><ref id="bib38"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Mitrea</surname><given-names>DM</given-names></name><name><surname>Mittasch</surname><given-names>M</given-names></name><name><surname>Gomes</surname><given-names>BF</given-names></name><name><surname>Klein</surname><given-names>IA</given-names></name><name><surname>Murcko</surname><given-names>MA</given-names></name></person-group><year iso-8601-date="2022">2022</year><article-title>Modulating biomolecular condensates: a novel approach to drug discovery</article-title><source>Nature Reviews. Drug Discovery</source><volume>21</volume><fpage>841</fpage><lpage>862</lpage><pub-id pub-id-type="doi">10.1038/s41573-022-00505-4</pub-id><pub-id pub-id-type="pmid">35974095</pub-id></element-citation></ref><ref id="bib39"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Moore</surname><given-names>JE</given-names></name><name><surname>Purcaro</surname><given-names>MJ</given-names></name><name><surname>Pratt</surname><given-names>HE</given-names></name><name><surname>Epstein</surname><given-names>CB</given-names></name><name><surname>Shoresh</surname><given-names>N</given-names></name><name><surname>Adrian</surname><given-names>J</given-names></name><name><surname>Kawli</surname><given-names>T</given-names></name><name><surname>Davis</surname><given-names>CA</given-names></name><name><surname>Dobin</surname><given-names>A</given-names></name><name><surname>Kaul</surname><given-names>R</given-names></name><name><surname>Halow</surname><given-names>J</given-names></name><name><surname>Van Nostrand</surname><given-names>EL</given-names></name><name><surname>Freese</surname><given-names>P</given-names></name><name><surname>Gorkin</surname><given-names>DU</given-names></name><name><surname>Shen</surname><given-names>Y</given-names></name><name><surname>He</surname><given-names>Y</given-names></name><name><surname>Mackiewicz</surname><given-names>M</given-names></name><name><surname>Pauli-Behn</surname><given-names>F</given-names></name><name><surname>Williams</surname><given-names>BA</given-names></name><name><surname>Mortazavi</surname><given-names>A</given-names></name><name><surname>Keller</surname><given-names>CA</given-names></name><name><surname>Zhang</surname><given-names>XO</given-names></name><name><surname>Elhajjajy</surname><given-names>SI</given-names></name><name><surname>Huey</surname><given-names>J</given-names></name><name><surname>Dickel</surname><given-names>DE</given-names></name><name><surname>Snetkova</surname><given-names>V</given-names></name><name><surname>Wei</surname><given-names>X</given-names></name><name><surname>Wang</surname><given-names>X</given-names></name><name><surname>Rivera-Mulia</surname><given-names>JC</given-names></name><name><surname>Rozowsky</surname><given-names>J</given-names></name><name><surname>Zhang</surname><given-names>J</given-names></name><name><surname>Chhetri</surname><given-names>SB</given-names></name><name><surname>Zhang</surname><given-names>J</given-names></name><name><surname>Victorsen</surname><given-names>A</given-names></name><name><surname>White</surname><given-names>KP</given-names></name><name><surname>Visel</surname><given-names>A</given-names></name><name><surname>Yeo</surname><given-names>GW</given-names></name><name><surname>Burge</surname><given-names>CB</given-names></name><name><surname>Lécuyer</surname><given-names>E</given-names></name><name><surname>Gilbert</surname><given-names>DM</given-names></name><name><surname>Dekker</surname><given-names>J</given-names></name><name><surname>Rinn</surname><given-names>J</given-names></name><name><surname>Mendenhall</surname><given-names>EM</given-names></name><name><surname>Ecker</surname><given-names>JR</given-names></name><name><surname>Kellis</surname><given-names>M</given-names></name><name><surname>Klein</surname><given-names>RJ</given-names></name><name><surname>Noble</surname><given-names>WS</given-names></name><name><surname>Kundaje</surname><given-names>A</given-names></name><name><surname>Guigó</surname><given-names>R</given-names></name><name><surname>Farnham</surname><given-names>PJ</given-names></name><name><surname>Cherry</surname><given-names>JM</given-names></name><name><surname>Myers</surname><given-names>RM</given-names></name><name><surname>Ren</surname><given-names>B</given-names></name><name><surname>Graveley</surname><given-names>BR</given-names></name><name><surname>Gerstein</surname><given-names>MB</given-names></name><name><surname>Pennacchio</surname><given-names>LA</given-names></name><name><surname>Snyder</surname><given-names>MP</given-names></name><name><surname>Bernstein</surname><given-names>BE</given-names></name><name><surname>Wold</surname><given-names>B</given-names></name><name><surname>Hardison</surname><given-names>RC</given-names></name><name><surname>Gingeras</surname><given-names>TR</given-names></name><name><surname>Stamatoyannopoulos</surname><given-names>JA</given-names></name><name><surname>Weng</surname><given-names>Z</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>Expanded encyclopaedias of DNA elements in the human and mouse genomes</article-title><source>Nature</source><volume>583</volume><fpage>699</fpage><lpage>710</lpage><pub-id pub-id-type="doi">10.1038/s41586-020-2493-4</pub-id><pub-id pub-id-type="pmid">32728249</pub-id></element-citation></ref><ref id="bib40"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Moorman</surname><given-names>C</given-names></name><name><surname>Sun</surname><given-names>LV</given-names></name><name><surname>Wang</surname><given-names>J</given-names></name><name><surname>de Wit</surname><given-names>E</given-names></name><name><surname>Talhout</surname><given-names>W</given-names></name><name><surname>Ward</surname><given-names>LD</given-names></name><name><surname>Greil</surname><given-names>F</given-names></name><name><surname>Lu</surname><given-names>XJ</given-names></name><name><surname>White</surname><given-names>KP</given-names></name><name><surname>Bussemaker</surname><given-names>HJ</given-names></name><name><surname>van Steensel</surname><given-names>B</given-names></name></person-group><year iso-8601-date="2006">2006</year><article-title>Hotspots of transcription factor colocalization in the genome of <italic>Drosophila melanogaster</italic></article-title><source>PNAS</source><volume>103</volume><fpage>12027</fpage><lpage>12032</lpage><pub-id pub-id-type="doi">10.1073/pnas.0605003103</pub-id><pub-id pub-id-type="pmid">16880385</pub-id></element-citation></ref><ref id="bib41"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Nair</surname><given-names>SJ</given-names></name><name><surname>Yang</surname><given-names>L</given-names></name><name><surname>Meluzzi</surname><given-names>D</given-names></name><name><surname>Oh</surname><given-names>S</given-names></name><name><surname>Yang</surname><given-names>F</given-names></name><name><surname>Friedman</surname><given-names>MJ</given-names></name><name><surname>Wang</surname><given-names>S</given-names></name><name><surname>Suter</surname><given-names>T</given-names></name><name><surname>Alshareedah</surname><given-names>I</given-names></name><name><surname>Gamliel</surname><given-names>A</given-names></name><name><surname>Ma</surname><given-names>Q</given-names></name><name><surname>Zhang</surname><given-names>J</given-names></name><name><surname>Hu</surname><given-names>Y</given-names></name><name><surname>Tan</surname><given-names>Y</given-names></name><name><surname>Ohgi</surname><given-names>KA</given-names></name><name><surname>Jayani</surname><given-names>RS</given-names></name><name><surname>Banerjee</surname><given-names>PR</given-names></name><name><surname>Aggarwal</surname><given-names>AK</given-names></name><name><surname>Rosenfeld</surname><given-names>MG</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Phase separation of ligand-activated enhancers licenses cooperative chromosomal enhancer assembly</article-title><source>Nature Structural &amp; Molecular Biology</source><volume>26</volume><fpage>193</fpage><lpage>203</lpage><pub-id pub-id-type="doi">10.1038/s41594-019-0190-5</pub-id><pub-id pub-id-type="pmid">30833784</pub-id></element-citation></ref><ref id="bib42"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Pachano</surname><given-names>T</given-names></name><name><surname>Sánchez-Gaya</surname><given-names>V</given-names></name><name><surname>Ealo</surname><given-names>T</given-names></name><name><surname>Mariner-Faulí</surname><given-names>M</given-names></name><name><surname>Bleckwehl</surname><given-names>T</given-names></name><name><surname>Asenjo</surname><given-names>HG</given-names></name><name><surname>Respuela</surname><given-names>P</given-names></name><name><surname>Cruz-Molina</surname><given-names>S</given-names></name><name><surname>Muñoz-San Martín</surname><given-names>M</given-names></name><name><surname>Haro</surname><given-names>E</given-names></name><name><surname>van IJcken</surname><given-names>WFJ</given-names></name><name><surname>Landeira</surname><given-names>D</given-names></name><name><surname>Rada-Iglesias</surname><given-names>A</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>Orphan CpG islands amplify poised enhancer regulatory activity and determine target gene responsiveness</article-title><source>Nature Genetics</source><volume>53</volume><fpage>1036</fpage><lpage>1049</lpage><pub-id pub-id-type="doi">10.1038/s41588-021-00888-x</pub-id><pub-id pub-id-type="pmid">34183853</pub-id></element-citation></ref><ref id="bib43"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Palacio</surname><given-names>M</given-names></name><name><surname>Taatjes</surname><given-names>DJ</given-names></name></person-group><year iso-8601-date="2022">2022</year><article-title>Merging established mechanisms with new insights: condensates, hubs, and the regulation of rna polymerase ii transcription</article-title><source>Journal of Molecular Biology</source><volume>434</volume><elocation-id>167216</elocation-id><pub-id pub-id-type="doi">10.1016/j.jmb.2021.167216</pub-id><pub-id pub-id-type="pmid">34474085</pub-id></element-citation></ref><ref id="bib44"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Palmer</surname><given-names>D</given-names></name><name><surname>Fabris</surname><given-names>F</given-names></name><name><surname>Doherty</surname><given-names>A</given-names></name><name><surname>Freitas</surname><given-names>AA</given-names></name><name><surname>de Magalhães</surname><given-names>JP</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>Ageing transcriptome meta-analysis reveals similarities and differences between key mammalian tissues</article-title><source>Aging</source><volume>13</volume><fpage>3313</fpage><lpage>3341</lpage><pub-id pub-id-type="doi">10.18632/aging.202648</pub-id><pub-id pub-id-type="pmid">33611312</pub-id></element-citation></ref><ref id="bib45"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Partridge</surname><given-names>EC</given-names></name><name><surname>Chhetri</surname><given-names>SB</given-names></name><name><surname>Prokop</surname><given-names>JW</given-names></name><name><surname>Ramaker</surname><given-names>RC</given-names></name><name><surname>Jansen</surname><given-names>CS</given-names></name><name><surname>Goh</surname><given-names>S-T</given-names></name><name><surname>Mackiewicz</surname><given-names>M</given-names></name><name><surname>Newberry</surname><given-names>KM</given-names></name><name><surname>Brandsmeier</surname><given-names>LA</given-names></name><name><surname>Meadows</surname><given-names>SK</given-names></name><name><surname>Messer</surname><given-names>CL</given-names></name><name><surname>Hardigan</surname><given-names>AA</given-names></name><name><surname>Coppola</surname><given-names>CJ</given-names></name><name><surname>Dean</surname><given-names>EC</given-names></name><name><surname>Jiang</surname><given-names>S</given-names></name><name><surname>Savic</surname><given-names>D</given-names></name><name><surname>Mortazavi</surname><given-names>A</given-names></name><name><surname>Wold</surname><given-names>BJ</given-names></name><name><surname>Myers</surname><given-names>RM</given-names></name><name><surname>Mendenhall</surname><given-names>EM</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>Occupancy maps of 208 chromatin-associated proteins in one human cell type</article-title><source>Nature</source><volume>583</volume><fpage>720</fpage><lpage>728</lpage><pub-id pub-id-type="doi">10.1038/s41586-020-2023-4</pub-id><pub-id pub-id-type="pmid">32728244</pub-id></element-citation></ref><ref id="bib46"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Quinlan</surname><given-names>AR</given-names></name><name><surname>Hall</surname><given-names>IM</given-names></name></person-group><year iso-8601-date="2010">2010</year><article-title>BEDTools: a flexible suite of utilities for comparing genomic features</article-title><source>Bioinformatics</source><volume>26</volume><fpage>841</fpage><lpage>842</lpage><pub-id pub-id-type="doi">10.1093/bioinformatics/btq033</pub-id><pub-id pub-id-type="pmid">20110278</pub-id></element-citation></ref><ref id="bib47"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Quinodoz</surname><given-names>SA</given-names></name><name><surname>Ollikainen</surname><given-names>N</given-names></name><name><surname>Tabak</surname><given-names>B</given-names></name><name><surname>Palla</surname><given-names>A</given-names></name><name><surname>Schmidt</surname><given-names>JM</given-names></name><name><surname>Detmar</surname><given-names>E</given-names></name><name><surname>Lai</surname><given-names>MM</given-names></name><name><surname>Shishkin</surname><given-names>AA</given-names></name><name><surname>Bhat</surname><given-names>P</given-names></name><name><surname>Takei</surname><given-names>Y</given-names></name><name><surname>Trinh</surname><given-names>V</given-names></name><name><surname>Aznauryan</surname><given-names>E</given-names></name><name><surname>Russell</surname><given-names>P</given-names></name><name><surname>Cheng</surname><given-names>C</given-names></name><name><surname>Jovanovic</surname><given-names>M</given-names></name><name><surname>Chow</surname><given-names>A</given-names></name><name><surname>Cai</surname><given-names>L</given-names></name><name><surname>McDonel</surname><given-names>P</given-names></name><name><surname>Garber</surname><given-names>M</given-names></name><name><surname>Guttman</surname><given-names>M</given-names></name></person-group><year iso-8601-date="2018">2018</year><article-title>Higher-order inter-chromosomal hubs shape 3d genome organization in the nucleus</article-title><source>Cell</source><volume>174</volume><fpage>744</fpage><lpage>757</lpage><pub-id pub-id-type="doi">10.1016/j.cell.2018.05.024</pub-id><pub-id pub-id-type="pmid">29887377</pub-id></element-citation></ref><ref id="bib48"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Ramaker</surname><given-names>RC</given-names></name><name><surname>Hardigan</surname><given-names>AA</given-names></name><name><surname>Goh</surname><given-names>ST</given-names></name><name><surname>Partridge</surname><given-names>EC</given-names></name><name><surname>Wold</surname><given-names>B</given-names></name><name><surname>Cooper</surname><given-names>SJ</given-names></name><name><surname>Myers</surname><given-names>RM</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>Dissecting the regulatory activity and sequence content of loci with exceptional numbers of transcription factor associations</article-title><source>Genome Research</source><volume>30</volume><fpage>939</fpage><lpage>950</lpage><pub-id pub-id-type="doi">10.1101/gr.260463.119</pub-id><pub-id pub-id-type="pmid">32616518</pub-id></element-citation></ref><ref id="bib49"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Rippe</surname><given-names>K</given-names></name><name><surname>Papantonis</surname><given-names>A</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>RNA polymerase II transcription compartments: from multivalent chromatin binding to liquid droplet formation?</article-title><source>Nature Reviews. Molecular Cell Biology</source><volume>22</volume><fpage>645</fpage><lpage>646</lpage><pub-id pub-id-type="doi">10.1038/s41580-021-00401-6</pub-id><pub-id pub-id-type="pmid">34282323</pub-id></element-citation></ref><ref id="bib50"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Rostam</surname><given-names>N</given-names></name><name><surname>Ghosh</surname><given-names>S</given-names></name><name><surname>Chow</surname><given-names>CFW</given-names></name><name><surname>Hadarovich</surname><given-names>A</given-names></name><name><surname>Landerer</surname><given-names>C</given-names></name><name><surname>Ghosh</surname><given-names>R</given-names></name><name><surname>Moon</surname><given-names>H</given-names></name><name><surname>Hersemann</surname><given-names>L</given-names></name><name><surname>Mitrea</surname><given-names>DM</given-names></name><name><surname>Klein</surname><given-names>IA</given-names></name><name><surname>Hyman</surname><given-names>AA</given-names></name><name><surname>Toth-Petroczy</surname><given-names>A</given-names></name></person-group><year iso-8601-date="2023">2023</year><article-title>CD-CODE: crowdsourcing condensate database and encyclopedia</article-title><source>Nature Methods</source><volume>20</volume><fpage>673</fpage><lpage>676</lpage><pub-id pub-id-type="doi">10.1038/s41592-023-01831-0</pub-id><pub-id pub-id-type="pmid">37024650</pub-id></element-citation></ref><ref id="bib51"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Roy</surname><given-names>S</given-names></name><name><surname>Ernst</surname><given-names>J</given-names></name><name><surname>Kharchenko</surname><given-names>PV</given-names></name><name><surname>Kheradpour</surname><given-names>P</given-names></name><name><surname>Negre</surname><given-names>N</given-names></name><name><surname>Eaton</surname><given-names>ML</given-names></name><name><surname>Landolin</surname><given-names>JM</given-names></name><name><surname>Bristow</surname><given-names>CA</given-names></name><name><surname>Ma</surname><given-names>L</given-names></name><name><surname>Lin</surname><given-names>MF</given-names></name><name><surname>Washietl</surname><given-names>S</given-names></name><name><surname>Arshinoff</surname><given-names>BI</given-names></name><name><surname>Ay</surname><given-names>F</given-names></name><name><surname>Meyer</surname><given-names>PE</given-names></name><name><surname>Robine</surname><given-names>N</given-names></name><name><surname>Washington</surname><given-names>NL</given-names></name><name><surname>Di Stefano</surname><given-names>L</given-names></name><name><surname>Berezikov</surname><given-names>E</given-names></name><name><surname>Brown</surname><given-names>CD</given-names></name><name><surname>Candeias</surname><given-names>R</given-names></name><name><surname>Carlson</surname><given-names>JW</given-names></name><name><surname>Carr</surname><given-names>A</given-names></name><name><surname>Jungreis</surname><given-names>I</given-names></name><name><surname>Marbach</surname><given-names>D</given-names></name><name><surname>Sealfon</surname><given-names>R</given-names></name><name><surname>Tolstorukov</surname><given-names>MY</given-names></name><name><surname>Will</surname><given-names>S</given-names></name><name><surname>Alekseyenko</surname><given-names>AA</given-names></name><name><surname>Artieri</surname><given-names>C</given-names></name><name><surname>Booth</surname><given-names>BW</given-names></name><name><surname>Brooks</surname><given-names>AN</given-names></name><name><surname>Dai</surname><given-names>Q</given-names></name><name><surname>Davis</surname><given-names>CA</given-names></name><name><surname>Duff</surname><given-names>MO</given-names></name><name><surname>Feng</surname><given-names>X</given-names></name><name><surname>Gorchakov</surname><given-names>AA</given-names></name><name><surname>Gu</surname><given-names>T</given-names></name><name><surname>Henikoff</surname><given-names>JG</given-names></name><name><surname>Kapranov</surname><given-names>P</given-names></name><name><surname>Li</surname><given-names>R</given-names></name><name><surname>MacAlpine</surname><given-names>HK</given-names></name><name><surname>Malone</surname><given-names>J</given-names></name><name><surname>Minoda</surname><given-names>A</given-names></name><name><surname>Nordman</surname><given-names>J</given-names></name><name><surname>Okamura</surname><given-names>K</given-names></name><name><surname>Perry</surname><given-names>M</given-names></name><name><surname>Powell</surname><given-names>SK</given-names></name><name><surname>Riddle</surname><given-names>NC</given-names></name><name><surname>Sakai</surname><given-names>A</given-names></name><name><surname>Samsonova</surname><given-names>A</given-names></name><name><surname>Sandler</surname><given-names>JE</given-names></name><name><surname>Schwartz</surname><given-names>YB</given-names></name><name><surname>Sher</surname><given-names>N</given-names></name><name><surname>Spokony</surname><given-names>R</given-names></name><name><surname>Sturgill</surname><given-names>D</given-names></name><name><surname>van Baren</surname><given-names>M</given-names></name><name><surname>Wan</surname><given-names>KH</given-names></name><name><surname>Yang</surname><given-names>L</given-names></name><name><surname>Yu</surname><given-names>C</given-names></name><name><surname>Feingold</surname><given-names>E</given-names></name><name><surname>Good</surname><given-names>P</given-names></name><name><surname>Guyer</surname><given-names>M</given-names></name><name><surname>Lowdon</surname><given-names>R</given-names></name><name><surname>Ahmad</surname><given-names>K</given-names></name><name><surname>Andrews</surname><given-names>J</given-names></name><name><surname>Berger</surname><given-names>B</given-names></name><name><surname>Brenner</surname><given-names>SE</given-names></name><name><surname>Brent</surname><given-names>MR</given-names></name><name><surname>Cherbas</surname><given-names>L</given-names></name><name><surname>Elgin</surname><given-names>SCR</given-names></name><name><surname>Gingeras</surname><given-names>TR</given-names></name><name><surname>Grossman</surname><given-names>R</given-names></name><name><surname>Hoskins</surname><given-names>RA</given-names></name><name><surname>Kaufman</surname><given-names>TC</given-names></name><name><surname>Kent</surname><given-names>W</given-names></name><name><surname>Kuroda</surname><given-names>MI</given-names></name><name><surname>Orr-Weaver</surname><given-names>T</given-names></name><name><surname>Perrimon</surname><given-names>N</given-names></name><name><surname>Pirrotta</surname><given-names>V</given-names></name><name><surname>Posakony</surname><given-names>JW</given-names></name><name><surname>Ren</surname><given-names>B</given-names></name><name><surname>Russell</surname><given-names>S</given-names></name><name><surname>Cherbas</surname><given-names>P</given-names></name><name><surname>Graveley</surname><given-names>BR</given-names></name><name><surname>Lewis</surname><given-names>S</given-names></name><name><surname>Micklem</surname><given-names>G</given-names></name><name><surname>Oliver</surname><given-names>B</given-names></name><name><surname>Park</surname><given-names>PJ</given-names></name><name><surname>Celniker</surname><given-names>SE</given-names></name><name><surname>Henikoff</surname><given-names>S</given-names></name><name><surname>Karpen</surname><given-names>GH</given-names></name><name><surname>Lai</surname><given-names>EC</given-names></name><name><surname>MacAlpine</surname><given-names>DM</given-names></name><name><surname>Stein</surname><given-names>LD</given-names></name><name><surname>White</surname><given-names>KP</given-names></name><name><surname>Kellis</surname><given-names>M</given-names></name><name><surname>Acevedo</surname><given-names>D</given-names></name><name><surname>Auburn</surname><given-names>R</given-names></name><name><surname>Barber</surname><given-names>G</given-names></name><name><surname>Bellen</surname><given-names>HJ</given-names></name><name><surname>Bishop</surname><given-names>EP</given-names></name><name><surname>Bryson</surname><given-names>TD</given-names></name><name><surname>Chateigner</surname><given-names>A</given-names></name><name><surname>Chen</surname><given-names>J</given-names></name><name><surname>Clawson</surname><given-names>H</given-names></name><name><surname>Comstock</surname><given-names>CLG</given-names></name><name><surname>Contrino</surname><given-names>S</given-names></name><name><surname>DeNapoli</surname><given-names>LC</given-names></name><name><surname>Ding</surname><given-names>Q</given-names></name><name><surname>Dobin</surname><given-names>A</given-names></name><name><surname>Domanus</surname><given-names>MH</given-names></name><name><surname>Drenkow</surname><given-names>J</given-names></name><name><surname>Dudoit</surname><given-names>S</given-names></name><name><surname>Dumais</surname><given-names>J</given-names></name><name><surname>Eng</surname><given-names>T</given-names></name><name><surname>Fagegaltier</surname><given-names>D</given-names></name><name><surname>Gadel</surname><given-names>SE</given-names></name><name><surname>Ghosh</surname><given-names>S</given-names></name><name><surname>Guillier</surname><given-names>F</given-names></name><name><surname>Hanley</surname><given-names>D</given-names></name><name><surname>Hannon</surname><given-names>GJ</given-names></name><name><surname>Hansen</surname><given-names>KD</given-names></name><name><surname>Heinz</surname><given-names>E</given-names></name><name><surname>Hinrichs</surname><given-names>AS</given-names></name><name><surname>Hirst</surname><given-names>M</given-names></name><name><surname>Jha</surname><given-names>S</given-names></name><name><surname>Jiang</surname><given-names>L</given-names></name><name><surname>Jung</surname><given-names>YL</given-names></name><name><surname>Kashevsky</surname><given-names>H</given-names></name><name><surname>Kennedy</surname><given-names>CD</given-names></name><name><surname>Kephart</surname><given-names>ET</given-names></name><name><surname>Langton</surname><given-names>L</given-names></name><name><surname>Lee</surname><given-names>O-K</given-names></name><name><surname>Li</surname><given-names>S</given-names></name><name><surname>Li</surname><given-names>Z</given-names></name><name><surname>Lin</surname><given-names>W</given-names></name><name><surname>Linder-Basso</surname><given-names>D</given-names></name><name><surname>Lloyd</surname><given-names>P</given-names></name><name><surname>Lyne</surname><given-names>R</given-names></name><name><surname>Marchetti</surname><given-names>SE</given-names></name><name><surname>Marra</surname><given-names>M</given-names></name><name><surname>Mattiuzzo</surname><given-names>NR</given-names></name><name><surname>McKay</surname><given-names>S</given-names></name><name><surname>Meyer</surname><given-names>F</given-names></name><name><surname>Miller</surname><given-names>D</given-names></name><name><surname>Miller</surname><given-names>SW</given-names></name><name><surname>Moore</surname><given-names>RA</given-names></name><name><surname>Morrison</surname><given-names>CA</given-names></name><name><surname>Prinz</surname><given-names>JA</given-names></name><name><surname>Rooks</surname><given-names>M</given-names></name><name><surname>Moore</surname><given-names>R</given-names></name><name><surname>Rutherford</surname><given-names>KM</given-names></name><name><surname>Ruzanov</surname><given-names>P</given-names></name><name><surname>Scheftner</surname><given-names>DA</given-names></name><name><surname>Senderowicz</surname><given-names>L</given-names></name><name><surname>Shah</surname><given-names>PK</given-names></name><name><surname>Shanower</surname><given-names>G</given-names></name><name><surname>Smith</surname><given-names>R</given-names></name><name><surname>Stinson</surname><given-names>EO</given-names></name><name><surname>Suchy</surname><given-names>S</given-names></name><name><surname>Tenney</surname><given-names>AE</given-names></name><name><surname>Tian</surname><given-names>F</given-names></name><name><surname>Venken</surname><given-names>KJT</given-names></name><name><surname>Wang</surname><given-names>H</given-names></name><name><surname>White</surname><given-names>R</given-names></name><name><surname>Wilkening</surname><given-names>J</given-names></name><name><surname>Willingham</surname><given-names>AT</given-names></name><name><surname>Zaleski</surname><given-names>C</given-names></name><name><surname>Zha</surname><given-names>Z</given-names></name><name><surname>Zhang</surname><given-names>D</given-names></name><name><surname>Zhao</surname><given-names>Y</given-names></name><name><surname>Zieba</surname><given-names>J</given-names></name><collab>The modENCODE Consortium</collab></person-group><year iso-8601-date="2010">2010</year><article-title>Identification of functional elements and regulatory circuits by <italic>Drosophila</italic> modENCODE</article-title><source>Science</source><volume>330</volume><fpage>1787</fpage><lpage>1797</lpage><pub-id pub-id-type="doi">10.1126/science.1198374</pub-id></element-citation></ref><ref id="bib52"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Schanze</surname><given-names>I</given-names></name><name><surname>Schanze</surname><given-names>D</given-names></name><name><surname>Bacino</surname><given-names>CA</given-names></name><name><surname>Douzgou</surname><given-names>S</given-names></name><name><surname>Kerr</surname><given-names>B</given-names></name><name><surname>Zenker</surname><given-names>M</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>Haploinsufficiency of SOX5, a member of the SOX (SRY-related HMG-box) family of transcription factors is a cause of intellectual disability</article-title><source>European Journal of Medical Genetics</source><volume>56</volume><fpage>108</fpage><lpage>113</lpage><pub-id pub-id-type="doi">10.1016/j.ejmg.2012.11.001</pub-id><pub-id pub-id-type="pmid">23220431</pub-id></element-citation></ref><ref id="bib53"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Schmitt</surname><given-names>AD</given-names></name><name><surname>Hu</surname><given-names>M</given-names></name><name><surname>Jung</surname><given-names>I</given-names></name><name><surname>Xu</surname><given-names>Z</given-names></name><name><surname>Qiu</surname><given-names>Y</given-names></name><name><surname>Tan</surname><given-names>CL</given-names></name><name><surname>Li</surname><given-names>Y</given-names></name><name><surname>Lin</surname><given-names>S</given-names></name><name><surname>Lin</surname><given-names>Y</given-names></name><name><surname>Barr</surname><given-names>CL</given-names></name><name><surname>Ren</surname><given-names>B</given-names></name></person-group><year iso-8601-date="2016">2016</year><article-title>A compendium of chromatin contact maps reveals spatially active regions in the human genome</article-title><source>Cell Reports</source><volume>17</volume><fpage>2042</fpage><lpage>2059</lpage><pub-id pub-id-type="doi">10.1016/j.celrep.2016.10.061</pub-id><pub-id pub-id-type="pmid">27851967</pub-id></element-citation></ref><ref id="bib54"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Serfling</surname><given-names>E</given-names></name><name><surname>Jasin</surname><given-names>M</given-names></name><name><surname>Schaffner</surname><given-names>W</given-names></name></person-group><year iso-8601-date="1985">1985</year><article-title>Enhancers and eukaryotic gene transcription</article-title><source>Trends in Genetics</source><volume>1</volume><fpage>224</fpage><lpage>230</lpage><pub-id pub-id-type="doi">10.1016/0168-9525(85)90088-5</pub-id></element-citation></ref><ref id="bib55"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Sethi</surname><given-names>A</given-names></name><name><surname>Gu</surname><given-names>M</given-names></name><name><surname>Gumusgoz</surname><given-names>E</given-names></name><name><surname>Chan</surname><given-names>L</given-names></name><name><surname>Yan</surname><given-names>K-K</given-names></name><name><surname>Rozowsky</surname><given-names>J</given-names></name><name><surname>Barozzi</surname><given-names>I</given-names></name><name><surname>Afzal</surname><given-names>V</given-names></name><name><surname>Akiyama</surname><given-names>JA</given-names></name><name><surname>Plajzer-Frick</surname><given-names>I</given-names></name><name><surname>Yan</surname><given-names>C</given-names></name><name><surname>Novak</surname><given-names>CS</given-names></name><name><surname>Kato</surname><given-names>M</given-names></name><name><surname>Garvin</surname><given-names>TH</given-names></name><name><surname>Pham</surname><given-names>Q</given-names></name><name><surname>Harrington</surname><given-names>A</given-names></name><name><surname>Mannion</surname><given-names>BJ</given-names></name><name><surname>Lee</surname><given-names>EA</given-names></name><name><surname>Fukuda-Yuzawa</surname><given-names>Y</given-names></name><name><surname>Visel</surname><given-names>A</given-names></name><name><surname>Dickel</surname><given-names>DE</given-names></name><name><surname>Yip</surname><given-names>KY</given-names></name><name><surname>Sutton</surname><given-names>R</given-names></name><name><surname>Pennacchio</surname><given-names>LA</given-names></name><name><surname>Gerstein</surname><given-names>M</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>Supervised enhancer prediction with epigenetic pattern recognition and targeted validation</article-title><source>Nature Methods</source><volume>17</volume><fpage>807</fpage><lpage>814</lpage><pub-id pub-id-type="doi">10.1038/s41592-020-0907-8</pub-id><pub-id pub-id-type="pmid">32737473</pub-id></element-citation></ref><ref id="bib56"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Shrinivas</surname><given-names>K</given-names></name><name><surname>Sabari</surname><given-names>BR</given-names></name><name><surname>Coffey</surname><given-names>EL</given-names></name><name><surname>Klein</surname><given-names>IA</given-names></name><name><surname>Boija</surname><given-names>A</given-names></name><name><surname>Zamudio</surname><given-names>AV</given-names></name><name><surname>Schuijers</surname><given-names>J</given-names></name><name><surname>Hannett</surname><given-names>NM</given-names></name><name><surname>Sharp</surname><given-names>PA</given-names></name><name><surname>Young</surname><given-names>RA</given-names></name><name><surname>Chakraborty</surname><given-names>AK</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>Enhancer features that drive formation of transcriptional condensates</article-title><source>Molecular Cell</source><volume>75</volume><fpage>549</fpage><lpage>561</lpage><pub-id pub-id-type="doi">10.1016/j.molcel.2019.07.009</pub-id><pub-id pub-id-type="pmid">31398323</pub-id></element-citation></ref><ref id="bib57"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Siepel</surname><given-names>A</given-names></name><name><surname>Bejerano</surname><given-names>G</given-names></name><name><surname>Pedersen</surname><given-names>JS</given-names></name><name><surname>Hinrichs</surname><given-names>AS</given-names></name><name><surname>Hou</surname><given-names>M</given-names></name><name><surname>Rosenbloom</surname><given-names>K</given-names></name><name><surname>Clawson</surname><given-names>H</given-names></name><name><surname>Spieth</surname><given-names>J</given-names></name><name><surname>Hillier</surname><given-names>LW</given-names></name><name><surname>Richards</surname><given-names>S</given-names></name><name><surname>Weinstock</surname><given-names>GM</given-names></name><name><surname>Wilson</surname><given-names>RK</given-names></name><name><surname>Gibbs</surname><given-names>RA</given-names></name><name><surname>Kent</surname><given-names>WJ</given-names></name><name><surname>Miller</surname><given-names>W</given-names></name><name><surname>Haussler</surname><given-names>D</given-names></name></person-group><year iso-8601-date="2005">2005</year><article-title>Evolutionarily conserved elements in vertebrate, insect, worm, and yeast genomes</article-title><source>Genome Research</source><volume>15</volume><fpage>1034</fpage><lpage>1050</lpage><pub-id pub-id-type="doi">10.1101/gr.3715005</pub-id><pub-id pub-id-type="pmid">16024819</pub-id></element-citation></ref><ref id="bib58"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Spitz</surname><given-names>F</given-names></name><name><surname>Furlong</surname><given-names>EEM</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>Transcription factors: from enhancer binding to developmental control</article-title><source>Nature Reviews. Genetics</source><volume>13</volume><fpage>613</fpage><lpage>626</lpage><pub-id pub-id-type="doi">10.1038/nrg3207</pub-id><pub-id pub-id-type="pmid">22868264</pub-id></element-citation></ref><ref id="bib59"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Szklarczyk</surname><given-names>D</given-names></name><name><surname>Gable</surname><given-names>AL</given-names></name><name><surname>Lyon</surname><given-names>D</given-names></name><name><surname>Junge</surname><given-names>A</given-names></name><name><surname>Wyder</surname><given-names>S</given-names></name><name><surname>Huerta-Cepas</surname><given-names>J</given-names></name><name><surname>Simonovic</surname><given-names>M</given-names></name><name><surname>Doncheva</surname><given-names>NT</given-names></name><name><surname>Morris</surname><given-names>JH</given-names></name><name><surname>Bork</surname><given-names>P</given-names></name><name><surname>Jensen</surname><given-names>LJ</given-names></name><name><surname>Von Mering</surname><given-names>C</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>STRING v11: protein-protein association networks with increased coverage, supporting functional discovery in genome-wide experimental datasets</article-title><source>Nucleic Acids Research</source><volume>47</volume><fpage>D607</fpage><lpage>D613</lpage><pub-id pub-id-type="doi">10.1093/nar/gky1131</pub-id><pub-id pub-id-type="pmid">30476243</pub-id></element-citation></ref><ref id="bib60"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Teytelman</surname><given-names>L</given-names></name><name><surname>Thurtle</surname><given-names>DM</given-names></name><name><surname>Rine</surname><given-names>J</given-names></name><name><surname>van Oudenaarden</surname><given-names>A</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>Highly expressed loci are vulnerable to misleading ChIP localization of multiple unrelated proteins</article-title><source>PNAS</source><volume>110</volume><fpage>18602</fpage><lpage>18607</lpage><pub-id pub-id-type="doi">10.1073/pnas.1316064110</pub-id><pub-id pub-id-type="pmid">24173036</pub-id></element-citation></ref><ref id="bib61"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Thanos</surname><given-names>D</given-names></name><name><surname>Maniatis</surname><given-names>T</given-names></name></person-group><year iso-8601-date="1995">1995</year><article-title>Virus induction of human IFN beta gene expression requires the assembly of an enhanceosome</article-title><source>Cell</source><volume>83</volume><fpage>1091</fpage><lpage>1100</lpage><pub-id pub-id-type="doi">10.1016/0092-8674(95)90136-1</pub-id><pub-id pub-id-type="pmid">8548797</pub-id></element-citation></ref><ref id="bib62"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>van Arensbergen</surname><given-names>J</given-names></name><name><surname>Pagie</surname><given-names>L</given-names></name><name><surname>FitzPatrick</surname><given-names>VD</given-names></name><name><surname>de Haas</surname><given-names>M</given-names></name><name><surname>Baltissen</surname><given-names>MP</given-names></name><name><surname>Comoglio</surname><given-names>F</given-names></name><name><surname>van der Weide</surname><given-names>RH</given-names></name><name><surname>Teunissen</surname><given-names>H</given-names></name><name><surname>Võsa</surname><given-names>U</given-names></name><name><surname>Franke</surname><given-names>L</given-names></name><name><surname>de Wit</surname><given-names>E</given-names></name><name><surname>Vermeulen</surname><given-names>M</given-names></name><name><surname>Bussemaker</surname><given-names>HJ</given-names></name><name><surname>van Steensel</surname><given-names>B</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>High-throughput identification of human SNPs affecting regulatory element activity</article-title><source>Nature Genetics</source><volume>51</volume><fpage>1160</fpage><lpage>1169</lpage><pub-id pub-id-type="doi">10.1038/s41588-019-0455-2</pub-id><pub-id pub-id-type="pmid">31253979</pub-id></element-citation></ref><ref id="bib63"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Vierstra</surname><given-names>J</given-names></name><name><surname>Lazar</surname><given-names>J</given-names></name><name><surname>Sandstrom</surname><given-names>R</given-names></name><name><surname>Halow</surname><given-names>J</given-names></name><name><surname>Lee</surname><given-names>K</given-names></name><name><surname>Bates</surname><given-names>D</given-names></name><name><surname>Diegel</surname><given-names>M</given-names></name><name><surname>Dunn</surname><given-names>D</given-names></name><name><surname>Neri</surname><given-names>F</given-names></name><name><surname>Haugen</surname><given-names>E</given-names></name><name><surname>Rynes</surname><given-names>E</given-names></name><name><surname>Reynolds</surname><given-names>A</given-names></name><name><surname>Nelson</surname><given-names>J</given-names></name><name><surname>Johnson</surname><given-names>A</given-names></name><name><surname>Frerker</surname><given-names>M</given-names></name><name><surname>Buckley</surname><given-names>M</given-names></name><name><surname>Kaul</surname><given-names>R</given-names></name><name><surname>Meuleman</surname><given-names>W</given-names></name><name><surname>Stamatoyannopoulos</surname><given-names>JA</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>Global reference mapping of human transcription factor footprints</article-title><source>Nature</source><volume>583</volume><fpage>729</fpage><lpage>736</lpage><pub-id pub-id-type="doi">10.1038/s41586-020-2528-x</pub-id><pub-id pub-id-type="pmid">32728250</pub-id></element-citation></ref><ref id="bib64"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Vinson</surname><given-names>C</given-names></name><name><surname>Chatterjee</surname><given-names>R</given-names></name><name><surname>Fitzgerald</surname><given-names>P</given-names></name></person-group><year iso-8601-date="2011">2011</year><article-title>Transcription factor binding sites and other features in human and <italic>Drosophila</italic> proximal promoters</article-title><source>Sub-Cellular Biochemistry</source><volume>52</volume><fpage>205</fpage><lpage>222</lpage><pub-id pub-id-type="doi">10.1007/978-90-481-9069-0_10</pub-id><pub-id pub-id-type="pmid">21557085</pub-id></element-citation></ref><ref id="bib65"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Virtanen</surname><given-names>P</given-names></name><name><surname>Gommers</surname><given-names>R</given-names></name><name><surname>Oliphant</surname><given-names>TE</given-names></name><name><surname>Haberland</surname><given-names>M</given-names></name><name><surname>Reddy</surname><given-names>T</given-names></name><name><surname>Cournapeau</surname><given-names>D</given-names></name><name><surname>Burovski</surname><given-names>E</given-names></name><name><surname>Peterson</surname><given-names>P</given-names></name><name><surname>Weckesser</surname><given-names>W</given-names></name><name><surname>Bright</surname><given-names>J</given-names></name><name><surname>van der Walt</surname><given-names>SJ</given-names></name><name><surname>Brett</surname><given-names>M</given-names></name><name><surname>Wilson</surname><given-names>J</given-names></name><name><surname>Millman</surname><given-names>KJ</given-names></name><name><surname>Mayorov</surname><given-names>N</given-names></name><name><surname>Nelson</surname><given-names>ARJ</given-names></name><name><surname>Jones</surname><given-names>E</given-names></name><name><surname>Kern</surname><given-names>R</given-names></name><name><surname>Larson</surname><given-names>E</given-names></name><name><surname>Carey</surname><given-names>CJ</given-names></name><name><surname>Polat</surname><given-names>İ</given-names></name><name><surname>Feng</surname><given-names>Y</given-names></name><name><surname>Moore</surname><given-names>EW</given-names></name><name><surname>VanderPlas</surname><given-names>J</given-names></name><name><surname>Laxalde</surname><given-names>D</given-names></name><name><surname>Perktold</surname><given-names>J</given-names></name><name><surname>Cimrman</surname><given-names>R</given-names></name><name><surname>Henriksen</surname><given-names>I</given-names></name><name><surname>Quintero</surname><given-names>EA</given-names></name><name><surname>Harris</surname><given-names>CR</given-names></name><name><surname>Archibald</surname><given-names>AM</given-names></name><name><surname>Ribeiro</surname><given-names>AH</given-names></name><name><surname>Pedregosa</surname><given-names>F</given-names></name><name><surname>van Mulbregt</surname><given-names>P</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>SciPy 1.0: fundamental algorithms for scientific computing in python</article-title><source>Nature Methods</source><volume>17</volume><fpage>261</fpage><lpage>272</lpage><pub-id pub-id-type="doi">10.1038/s41592-019-0686-2</pub-id><pub-id pub-id-type="pmid">32015543</pub-id></element-citation></ref><ref id="bib66"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Wang</surname><given-names>J</given-names></name><name><surname>Zhuang</surname><given-names>J</given-names></name><name><surname>Iyer</surname><given-names>S</given-names></name><name><surname>Lin</surname><given-names>XY</given-names></name><name><surname>Greven</surname><given-names>MC</given-names></name><name><surname>Kim</surname><given-names>BH</given-names></name><name><surname>Moore</surname><given-names>J</given-names></name><name><surname>Pierce</surname><given-names>BG</given-names></name><name><surname>Dong</surname><given-names>X</given-names></name><name><surname>Virgil</surname><given-names>D</given-names></name><name><surname>Birney</surname><given-names>E</given-names></name><name><surname>Hung</surname><given-names>JH</given-names></name><name><surname>Weng</surname><given-names>Z</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>Factorbook.org: a Wiki-based database for transcription factor-binding data generated by the ENCODE consortium</article-title><source>Nucleic Acids Research</source><volume>41</volume><fpage>D171</fpage><lpage>D176</lpage><pub-id pub-id-type="doi">10.1093/nar/gks1221</pub-id><pub-id pub-id-type="pmid">23203885</pub-id></element-citation></ref><ref id="bib67"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Waskom</surname><given-names>M</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>seaborn: statistical data visualization</article-title><source>Journal of Open Source Software</source><volume>6</volume><elocation-id>3021</elocation-id><pub-id pub-id-type="doi">10.21105/joss.03021</pub-id></element-citation></ref><ref id="bib68"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Wei</surname><given-names>MT</given-names></name><name><surname>Chang</surname><given-names>YC</given-names></name><name><surname>Shimobayashi</surname><given-names>SF</given-names></name><name><surname>Shin</surname><given-names>Y</given-names></name><name><surname>Strom</surname><given-names>AR</given-names></name><name><surname>Brangwynne</surname><given-names>CP</given-names></name></person-group><year iso-8601-date="2020">2020</year><article-title>Nucleated transcriptional condensates amplify gene expression</article-title><source>Nature Cell Biology</source><volume>22</volume><fpage>1187</fpage><lpage>1196</lpage><pub-id pub-id-type="doi">10.1038/s41556-020-00578-6</pub-id><pub-id pub-id-type="pmid">32929202</pub-id></element-citation></ref><ref id="bib69"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>White</surname><given-names>SM</given-names></name><name><surname>Snyder</surname><given-names>MP</given-names></name><name><surname>Yi</surname><given-names>C</given-names></name></person-group><year iso-8601-date="2021">2021</year><article-title>Master lineage transcription factors anchor trans mega transcriptional complexes at highly accessible enhancer sites to promote long-range chromatin clustering and transcription of distal target genes</article-title><source>Nucleic Acids Research</source><volume>49</volume><fpage>12196</fpage><lpage>12210</lpage><pub-id pub-id-type="doi">10.1093/nar/gkab1105</pub-id><pub-id pub-id-type="pmid">34850122</pub-id></element-citation></ref><ref id="bib70"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Whyte</surname><given-names>WA</given-names></name><name><surname>Orlando</surname><given-names>DA</given-names></name><name><surname>Hnisz</surname><given-names>D</given-names></name><name><surname>Abraham</surname><given-names>BJ</given-names></name><name><surname>Lin</surname><given-names>CY</given-names></name><name><surname>Kagey</surname><given-names>MH</given-names></name><name><surname>Rahl</surname><given-names>PB</given-names></name><name><surname>Lee</surname><given-names>TI</given-names></name><name><surname>Young</surname><given-names>RA</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>Master transcription factors and mediator establish super-enhancers at key cell identity genes</article-title><source>Cell</source><volume>153</volume><fpage>307</fpage><lpage>319</lpage><pub-id pub-id-type="doi">10.1016/j.cell.2013.03.035</pub-id><pub-id pub-id-type="pmid">23582322</pub-id></element-citation></ref><ref id="bib71"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Wreczycka</surname><given-names>K</given-names></name><name><surname>Franke</surname><given-names>V</given-names></name><name><surname>Uyar</surname><given-names>B</given-names></name><name><surname>Wurmus</surname><given-names>R</given-names></name><name><surname>Bulut</surname><given-names>S</given-names></name><name><surname>Tursun</surname><given-names>B</given-names></name><name><surname>Akalin</surname><given-names>A</given-names></name></person-group><year iso-8601-date="2019">2019</year><article-title>HOT or not: examining the basis of high-occupancy target regions</article-title><source>Nucleic Acids Research</source><volume>47</volume><fpage>5735</fpage><lpage>5745</lpage><pub-id pub-id-type="doi">10.1093/nar/gkz460</pub-id><pub-id pub-id-type="pmid">31114922</pub-id></element-citation></ref><ref id="bib72"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Wunderlich</surname><given-names>Z</given-names></name><name><surname>Mirny</surname><given-names>LA</given-names></name></person-group><year iso-8601-date="2009">2009</year><article-title>Different gene regulation strategies revealed by analysis of binding motifs</article-title><source>Trends in Genetics</source><volume>25</volume><fpage>434</fpage><lpage>440</lpage><pub-id pub-id-type="doi">10.1016/j.tig.2009.08.003</pub-id><pub-id pub-id-type="pmid">19815308</pub-id></element-citation></ref><ref id="bib73"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Xie</surname><given-names>D</given-names></name><name><surname>Boyle</surname><given-names>AP</given-names></name><name><surname>Wu</surname><given-names>L</given-names></name><name><surname>Zhai</surname><given-names>J</given-names></name><name><surname>Kawli</surname><given-names>T</given-names></name><name><surname>Snyder</surname><given-names>M</given-names></name></person-group><year iso-8601-date="2013">2013</year><article-title>Dynamic trans-acting factor colocalization in human cells</article-title><source>Cell</source><volume>155</volume><fpage>713</fpage><lpage>724</lpage><pub-id pub-id-type="doi">10.1016/j.cell.2013.09.043</pub-id><pub-id pub-id-type="pmid">24243024</pub-id></element-citation></ref><ref id="bib74"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Yao</surname><given-names>L</given-names></name><name><surname>Liang</surname><given-names>J</given-names></name><name><surname>Ozer</surname><given-names>A</given-names></name><name><surname>Leung</surname><given-names>AKY</given-names></name><name><surname>Lis</surname><given-names>JT</given-names></name><name><surname>Yu</surname><given-names>H</given-names></name></person-group><year iso-8601-date="2022">2022</year><article-title>A comparison of experimental assays and analytical methods for genome-wide identification of active enhancers</article-title><source>Nature Biotechnology</source><volume>40</volume><fpage>1056</fpage><lpage>1065</lpage><pub-id pub-id-type="doi">10.1038/s41587-022-01211-7</pub-id><pub-id pub-id-type="pmid">35177836</pub-id></element-citation></ref><ref id="bib75"><element-citation publication-type="journal"><person-group person-group-type="author"><name><surname>Yip</surname><given-names>KY</given-names></name><name><surname>Cheng</surname><given-names>C</given-names></name><name><surname>Bhardwaj</surname><given-names>N</given-names></name><name><surname>Brown</surname><given-names>JB</given-names></name><name><surname>Leng</surname><given-names>J</given-names></name><name><surname>Kundaje</surname><given-names>A</given-names></name><name><surname>Rozowsky</surname><given-names>J</given-names></name><name><surname>Birney</surname><given-names>E</given-names></name><name><surname>Bickel</surname><given-names>P</given-names></name><name><surname>Snyder</surname><given-names>M</given-names></name><name><surname>Gerstein</surname><given-names>M</given-names></name></person-group><year iso-8601-date="2012">2012</year><article-title>Classification of human genomic regions based on experimentally determined binding sites of more than 100 transcription-related factors</article-title><source>Genome Biology</source><volume>13</volume><elocation-id>R48</elocation-id><pub-id pub-id-type="doi">10.1186/gb-2012-13-9-r48</pub-id><pub-id pub-id-type="pmid">22950945</pub-id></element-citation></ref></ref-list><app-group><app id="appendix-1"><title>Appendix 1</title><sec sec-type="appendix" id="s8"><title>Joint DAPs analysis</title><p>To jointly analyze the conditional distributions of ChIP-seq signal levels in the presence/absence of individual DAPs, we extracted a square matrix of size n=545 using the DAPs present in HepG2. For each analyzed 400 bp locus, we extracted bound DAPs together with the ChIP-seq signal values. Then, each binary combination of the DAPs bound in that locus is added to their respective cells on the square matrix. Afterward, the matrix is normalized along the x-axis with the maximum value of each row. In other words, each row represents the normalized ChIP-seq signal strength in the presence of DAP indicated on the x-axis. We treated the empty values as 0 and removed the main diagonal values, leading to the matrix size of 544×544. Hierarchical clustering was done using the UPGMA algorithm.</p></sec><sec sec-type="appendix" id="s9"><title>Hi-C 3D chromatin analysis</title><p>Hi-C data analysis was carried out on HepG2. The datasets were obtained from ENCODE Project: ENCFF050EKS (chromatin loops), ENCFF018XKF (TADs), ENCFF548XLR (hic file). The coordinates in all of the datasets were converted from hg38 to hg19 using LiftOver.</p><p>The significant long-range contacts with 5 kb resolution were extracted using the FitHiChIP program (<xref ref-type="bibr" rid="bib5">Bhattacharyya et al., 2019</xref>) with a threshold q-value&lt;0.0001 and ICE bias correction, using <italic>all-against-all</italic> option.</p><p>The loci with &gt;50% overlap were considered for the analysis of loops, TADs, and long-range chromatin contacts. Using the long-range chromatin contacts, we constructed a graph such that each node is the analyzed 400 bp locus and the edge is a long-range chromatin contact if the two connected nodes are located on different legs of the chromatin contacts. Based on this graph, we calculated the total number of contacts between the loci located in different bins, leading to a 14×14 matrix. We then normalized the values in each cell of the matrix with the maximum number of contacts in all cells.</p><p>FIRE loci were extracted from the .<italic>hic</italic> file using FIRECaller R package (<xref ref-type="bibr" rid="bib53">Schmitt et al., 2016</xref>).</p><p>For enrichment analyses of all the mentioned Hi-C-related regions, the ATAC-seq regions were used as background.</p></sec><sec sec-type="appendix" id="s10"><title>PPI enrichment analysis</title><p>To test the significance of the PPI networks described above, we ran 100 trials for each cluster by randomly selecting an equal number of DAPs reported in PPI networks and calculated the significance of the PPI enrichment p-values. All of the reported PPI enrichment p-values were significantly higher than the randomized trials (p-value&lt;0.01, one-sample t-test).</p><p>PPI networks and PPI enrichment p-values were extracted using the STRING Database’s API (<ext-link ext-link-type="uri" xlink:href="https://string-db.org/cgi/help.pl?subpage=api">https://string-db.org/cgi/help.pl?subpage=api</ext-link>). For each cluster of DAPs analyzed, we submitted the list of DAPs as identifiers and retrieved the p-values using the <italic>ppi_enrichment</italic> interface. For each cluster, we extracted 100 PPI enrichment p-values each time randomly selecting DAPs in equal numbers to the size of the analyzed cluster. We then used the set of 100 p-values as a background distribution and conducted a one-sample t-test, where by the null hypothesis the p-value of the cluster is the mean of 100 p-values and computed the p-values of significance of the reported PPI network. The results of this analysis are in <xref ref-type="supplementary-material" rid="supp1">Supplementary file 1, table S2</xref>.</p></sec><sec sec-type="appendix" id="s11"><title>Classification results analyses</title><p>Sequence-based classification experiments were carried out using CNNs (one-hot encoded) and gkmSVM (gapped k-mers). For feature-based classification, we trained logistic regression (LogReg) classifiers and separate SVM classifiers using kernel functions of linear, polynomial, RBF, and sigmoid.</p><p>Using the sequence features, we trained separate models using each of the features in addition to one with all of the features combined. We observed that, when averaged across all the methods, GC content value possesses the highest amount of discrimination power (auROC: 0.73), followed by the combination of all features (auROC: 0.70) (<xref ref-type="fig" rid="app1fig1">Appendix 1—figure 1A</xref>). When compared across the classification methods, LogReg and SVM with linear kernel outperformed the other non-linear kernels by 20%, suggesting that the features possess linearly combined or largely overlapping effects in encoding the information in HOT loci (<xref ref-type="fig" rid="app1fig2">Appendix 1—figure 2B</xref>).</p><p>When classified using the sequences directly, CNN yielded the highest performance with auROC of 0.91, while for the gkmSVM it was 0.86 (both averaged over cell lines and control sets), suggesting that CNNs capture the motif grammar of the HOT loci better than gapped k-mers (<xref ref-type="fig" rid="app1fig2">Appendix 1—figure 2</xref>). When the two classification schemes (sequence- and feature-based) are compared, CNNs outperformed the LogReg and linear SVMs by a factor of 1.3× (or 17%).</p></sec><sec sec-type="appendix" id="s12"><title>Classification datasets</title><p>For the classification of HOT loci, three different setups were constructed using the control (negative) sets:</p><list list-type="simple"><list-item><p>- Randomly selected from the merged DHS regions obtained from the Roadmap Epigenomics Project to be 10× the size of the positive set (HOT loci)</p></list-item><list-item><p>- Regular enhancers (see Methods: Definitions), with the HOT loci subtracted</p></list-item><list-item><p>- Regular promoters (see Methods: Definitions)</p></list-item></list><p>The regions from chromosomes 6,7 were used as validation sets, chromosomes 8,9 were used as test sets, and the rest of the autosomal chromosomes were used as training sets.</p><p>The total number of regions in classification setups and their train/validation/test sets splits is as follows:</p><table-wrap id="inlinetable1" position="anchor"><table frame="hsides" rules="groups"><thead><tr><th align="left" valign="bottom" colspan="6">Controls: DHS</th></tr></thead><tbody><tr><td align="left" valign="bottom">Cell line</td><td align="left" valign="bottom">HOTs</td><td align="left" valign="bottom">Controls</td><td align="left" valign="bottom">Train</td><td align="left" valign="bottom">Validation</td><td align="left" valign="bottom">Test</td></tr><tr><td align="left" valign="bottom">HepG2</td><td align="char" char="." valign="bottom">25,928</td><td align="char" char="." valign="bottom">249,499</td><td align="char" char="." valign="bottom">210,520</td><td align="char" char="." valign="bottom">33,231</td><td align="char" char="." valign="bottom">31,676</td></tr><tr><td align="left" valign="bottom">K562</td><td align="char" char="." valign="bottom">15,231</td><td align="char" char="." valign="bottom">146,585</td><td align="char" char="." valign="bottom">123,041</td><td align="char" char="." valign="bottom">20,310</td><td align="char" char="." valign="bottom">18,465</td></tr><tr><td align="left" valign="bottom" colspan="6">Controls: regular enhancers</td></tr><tr><td align="left" valign="bottom">Cell line</td><td align="left" valign="bottom">HOTs</td><td align="left" valign="bottom">Controls</td><td align="left" valign="bottom">Train</td><td align="left" valign="bottom">Validation</td><td align="left" valign="bottom">Test</td></tr><tr><td align="left" valign="bottom">HepG2</td><td align="char" char="." valign="bottom">25,928</td><td align="char" char="." valign="bottom">249,499</td><td align="char" char="." valign="bottom">210,520</td><td align="char" char="." valign="bottom">33,231</td><td align="char" char="." valign="bottom">31,676</td></tr><tr><td align="left" valign="bottom">K562</td><td align="char" char="." valign="bottom">15,231</td><td align="char" char="." valign="bottom">146,585</td><td align="char" char="." valign="bottom">123,041</td><td align="char" char="." valign="bottom">20,310</td><td align="char" char="." valign="bottom">18,465</td></tr><tr><td align="left" valign="bottom" colspan="6">Controls: regular promoters</td></tr><tr><td align="left" valign="bottom">Cell line</td><td align="left" valign="bottom">HOTs</td><td align="left" valign="bottom">Controls</td><td align="left" valign="bottom">Train</td><td align="left" valign="bottom">Validation</td><td align="left" valign="bottom">Test</td></tr><tr><td align="left" valign="bottom">HepG2</td><td align="char" char="." valign="bottom">25,928</td><td align="char" char="." valign="bottom">28,621</td><td align="char" char="." valign="bottom">34,970</td><td align="char" char="." valign="bottom">5479</td><td align="char" char="." valign="bottom">3403</td></tr><tr><td align="left" valign="bottom">K562</td><td align="char" char="." valign="bottom">15,231</td><td align="char" char="." valign="bottom">25,810</td><td align="char" char="." valign="bottom">41,979</td><td align="char" char="." valign="bottom">5800</td><td align="char" char="." valign="bottom">3959</td></tr></tbody></table></table-wrap></sec><sec sec-type="appendix" id="s13"><title>Sequence-based classification</title><p>For training CNNs, the sequences of the loci were converted to one-hot encoding, with the lengths options of 400 bp and extended to 1000 bp.</p><p>The model consists of the layers as follows:</p><table-wrap id="inlinetable2" position="anchor"><table frame="hsides" rules="groups"><thead><tr><th align="left" valign="bottom">Layer</th><th align="left" valign="bottom">Params</th><th align="left" valign="bottom">Activation</th></tr></thead><tbody><tr><td align="left" valign="bottom">1. Convolutional</td><td align="left" valign="bottom">filters = 480, kernel_size = 9, stride = 1</td><td align="left" valign="bottom">ReLu</td></tr><tr><td align="left" valign="bottom">2. Max pool</td><td align="left" valign="bottom">Pool_size = 9, stride = 3</td><td align="left" valign="bottom"/></tr><tr><td align="left" valign="bottom">3. Droupout</td><td align="left" valign="bottom">p=0.2</td><td align="left" valign="bottom"/></tr><tr><td align="left" valign="bottom">4. Convolutional</td><td align="left" valign="bottom">filters = 480, kernel_size = 4, stride = 1</td><td align="left" valign="bottom">ReLu</td></tr><tr><td align="left" valign="bottom">5. Max pool</td><td align="left" valign="bottom">Pool_size = 4, stride = 2</td><td align="left" valign="bottom"/></tr><tr><td align="left" valign="bottom">6. Droupout</td><td align="left" valign="bottom">p=0.2</td><td align="left" valign="bottom"/></tr><tr><td align="left" valign="bottom">7. Convolutional</td><td align="left" valign="bottom">filters = 240, kernel_size = 4, stride = 1</td><td align="left" valign="bottom">ReLu</td></tr><tr><td align="left" valign="bottom">8. Max pool</td><td align="left" valign="bottom">Pool_size = 4, stride = 2</td><td align="left" valign="bottom"/></tr><tr><td align="left" valign="bottom">9. Droupout</td><td align="left" valign="bottom">p=0.2</td><td align="left" valign="bottom"/></tr><tr><td align="left" valign="bottom">10. Convolutional</td><td align="left" valign="bottom">filters = 320, kernel_size = 4, stride = 1</td><td align="left" valign="bottom">ReLu</td></tr><tr><td align="left" valign="bottom">11. Max pool</td><td align="left" valign="bottom">Pool_size = 4, stride = 2</td><td align="left" valign="bottom"/></tr><tr><td align="left" valign="bottom">12. Fully connected</td><td align="left" valign="bottom">units = 180</td><td align="left" valign="bottom">ReLu</td></tr><tr><td align="left" valign="bottom">13. Fully connected</td><td align="left" valign="bottom">units = 15</td><td align="left" valign="bottom">Sigmoid</td></tr></tbody></table></table-wrap><p>Total number of trainable parameters is 2,342,723. The kernels were subjected to constraints of max_norm = 0.9, l1=5*10E-7, l2=1E-8. Each instance of the model was trained using the input lengths of 400 and 1000. The training process was run for a maximum of 200 epochs with a patience period of 15. The models were built using <italic>tensorflow v2.3.1</italic> and trained on NVIDIA k80 GPUs.</p><p>For SVM classification, gapped k-mer SVM program was used and downloaded from <ext-link ext-link-type="uri" xlink:href="https://github.com/Dongwon-Lee/lsgkm">https://github.com/Dongwon-Lee/lsgkm</ext-link> (<xref ref-type="bibr" rid="bib30">Lee, 2023</xref>). For each category of the regions, instances of SVM models were trained, using 400 bp and 1000 bp regions, with the following kernel options:</p><list list-type="simple"><list-item><p>0 -- gapped-kmer</p></list-item><list-item><p>1 -- estimated l-mer with full filter</p></list-item><list-item><p>2 -- estimated l-mer with truncated filter (gkm)</p></list-item><list-item><p>3 -- gkm+RBF (gkmrbf)</p></list-item><list-item><p>4 -- gkm+center weighted (wgkm)</p></list-item><list-item><p>5 -- gkm+center weighted+RBF (wgkmrbf)</p></list-item></list></sec><sec sec-type="appendix" id="s14"><title>Feature-based classification</title><p>The features used for classification were:</p><list list-type="simple"><list-item><p>- GC content.</p></list-item><list-item><p>- CpG content: counted the occurrences of ‘CG’ as density over the sequence length.</p></list-item><list-item><p>- GpC content: counted the occurrences of ‘GC’ as density over the sequence length.</p></list-item><list-item><p>- CpG island coverage: fraction of the overlaps with the CpG island obtained from UCSC Genome Browser database.</p></list-item></list><p>Each classification model was trained using all of the features at once (n=4) and using each of the features separately.</p><sec sec-type="appendix" id="s14-1"><title>Logistic regression</title><p><italic>sklearn.linear_model.LogisticRegression</italic> API was used from scikit-learn library.</p></sec><sec sec-type="appendix" id="s14-2"><title>SVM</title><p>sklearn.svm.SVM API was used from scikit-learn library. Kernels used with SVM classification are <italic>linear</italic>, <italic>polynomial</italic>, <italic>radial basis function(rbf</italic>), and <italic>sigmoid</italic>.</p><fig id="app1fig1" position="float"><label>Appendix 1—figure 1.</label><caption><title>Classification of high-occupancy target (HOT) loci using the sequence features.</title><p>(<bold>A</bold>) Classification performances when each sequence feature (<italic>GC, GpC, CpG, CGI</italic>) is used separately and all of them simultaneously (<italic>all</italic>). Error bar variations across cell lines, classification methods, and control sets. (<bold>B</bold>) Classification performances of different methods. Error bar variations across cell lines, sequence features, and control sets.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-95170-app1-fig1-v1.tif"/></fig><fig id="app1fig2" position="float"><label>Appendix 1—figure 2.</label><caption><title>Classification of high-occupancy target (HOT) loci using the sequences directly.</title><p>Error bar variations across cell lines and control sets. See Appendix 1 – Sequence-based classification for details of methods used.</p></caption><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-95170-app1-fig2-v1.tif"/></fig></sec></sec></app></app-group></back><sub-article article-type="editor-report" id="sa0"><front-stub><article-id pub-id-type="doi">10.7554/eLife.95170.3.sa0</article-id><title-group><article-title>eLife Assessment</article-title></title-group><contrib-group><contrib contrib-type="author"><name><surname>Altemose</surname><given-names>Nicolas</given-names></name><role specific-use="editor">Reviewing Editor</role><aff><institution>Stanford University</institution><country>United States</country></aff></contrib></contrib-group><kwd-group kwd-group-type="evidence-strength"><kwd>Solid</kwd></kwd-group><kwd-group kwd-group-type="claim-importance"><kwd>Valuable</kwd></kwd-group></front-stub><body><p>This <bold>valuable</bold> study explores the sequence characteristics and conservation of high-occupancy target loci, regions in the human genome such as promoters and enhancers that are bound by a multitude of transcription factors. The computational analyses presented in this study are <bold>solid</bold>. This study would be a helpful resource for researchers performing ChIP-seq based analyses of transcription factor binding.</p></body></sub-article><sub-article article-type="referee-report" id="sa1"><front-stub><article-id pub-id-type="doi">10.7554/eLife.95170.3.sa1</article-id><title-group><article-title>Reviewer #1 (Public review):</article-title></title-group><contrib-group><contrib contrib-type="author"><anonymous/><role specific-use="referee">Reviewer</role></contrib></contrib-group></front-stub><body><p>Summary:</p><p>This study explores the sequence characteristics and features of high-occupancy target (HOT) loci across the human genome. The computational analyses presented in this paper provide information into the correlation of TF binding and regulatory networks at HOT loci that were regarded as lacking sequence specificity.</p><p>By leveraging hundreds of ChIP-seq datasets from the ENCODE Project to delineate HOT loci in HepG2, K562, and H1-hESC cells, the investigators identified the regulatory significance and participation in 3D chromatin interactions of HOT loci. Subsequent exploration focused on the interaction of DNA-associated proteins (DAPs) with HOT loci using computational models. The models established that the potential formation of HOT loci is likely embedded in their DNA sequences and is significantly influenced by GC contents. Further inquiry exposed contrasting roles of HOT loci in housekeeping and tissue-specific functions spanning various cell types, with distinctions between embryonic and differentiated states, including instances of polymorphic variability. The authors conclude with a speculative model that HOT loci serve as anchors where phase-separated transcriptional condensates form. The findings presented here open avenues for future research, encouraging more exploration of the functional implications of HOT loci.</p><p>Strengths:</p><p>The concept of using computational models to define characteristics of HOT loci is refreshing and allows researchers to take a different approach in identifying potential targets. The major strengths of the study lie in the very large number of datasets analyzed, with hundreds of ChIP-seq data sets for both HepG2 and K562 cells as part of the ENCODE project. Such quantitative power allowed the authors to delve deeply into HOT loci, which were previously thought to be artifacts.</p><p>Weaknesses:</p><p>While this study contributes to our knowledge of HOT loci, there are critical weaknesses that need to be addressed. There are questions on the validity of the assumptions made for certain analyses. The speculative nature of the proposed model involving transcriptional condensates needs either further validation or be toned down. Furthermore, some apparent contradictions exist among the main conclusions, and these either need to be better explained or corrected. Lastly, several figure panels could be better explained or described in the figure legends.</p><p>Update After Revisions:</p><p>The authors have addressed the above comments and concerns appropriately. The addition of the new Figure 9 is particularly compelling and strengthens the authors' conclusions. This reviewer has no further concerns.</p></body></sub-article><sub-article article-type="referee-report" id="sa2"><front-stub><article-id pub-id-type="doi">10.7554/eLife.95170.3.sa2</article-id><title-group><article-title>Reviewer #2 (Public review):</article-title></title-group><contrib-group><contrib contrib-type="author"><anonymous/><role specific-use="referee">Reviewer</role></contrib></contrib-group></front-stub><body><p>Summary:</p><p>The paper by Hydaiberdiev and Ovcharenko offers comprehensive analyses and insights about the 'high-occupancy target' (HOT) loci in the human genome. These are considered genomic regions that overlap with transcription factor binding sites. The authors provided very comprehensive analyses of the TF composition characteristics of these HOT loci. They showed that these HOT loci tend to overlap with annotated promoters and enhancers, GC-rich regions, open chromatin signals, and highly conserved regions and that these loci are also enriched with potentially causal variants with different traits.</p><p>Strengths:</p><p>Overall, the HOT loci' definition is clear and the data of HOT regions across the genome can be a useful dataset for studies that use HepG2 or K562 as a model. I appreciate the authors' efforts in presenting many analyses and plots backing up each statement.</p><p>Comments on revised version:</p><p>In the second round of review, I think the authors have sufficiently addressed all of my previous comments. The study itself is very comprehensive, tackling all aspects of the HOT loci, though I still find the paper to be unnecessarily long and long-winded. That said, being consistent with the long and detailed paper, the provided Github repository and Zenodo archive is well-documented. I appreciate that the authors include detailed readme about the different datafiles available for readers. The list of HOT loci is probably the most useful asset in this manuscript and the authors did a good job documenting data availability in both Github and Zenodo.</p></body></sub-article><sub-article article-type="referee-report" id="sa3"><front-stub><article-id pub-id-type="doi">10.7554/eLife.95170.3.sa3</article-id><title-group><article-title>Reviewer #3 (Public review):</article-title></title-group><contrib-group><contrib contrib-type="author"><anonymous/><role specific-use="referee">Reviewer</role></contrib></contrib-group></front-stub><body><p>Summary:</p><p>Hudaiberdiev and Ovcharenko investigate regions within the genome where a high abundance of DNA associated proteins are located and identify DNA sequence feature enriched in these regions, their conservation in evolution, and variation in disease. Using ChIP-seq binding profiles of over 1,000 proteins in three human cell lines (HepG2, K562, and H1) as a data source they're able to identify nearly 44,000 high-occupancy target loci (HOT) that form at promoter and enhancer regions, thus suggesting these HOT loci regulate housekeeping and cell identity genes. Their primary investigative tool is HepG2 cells, but they employ K562 and H1 cells as tools to validate these assertions in other human cell types. Their analyses use RNA pol II signal, super enhancer, regular enhancer and epigentic marks to support the identification of these regions. The work is notable, in that it identifies a set of proteins that are invariantly associated with high-occupancy enhancers and promoters and argues for the integration of these molecules at different genomic loci. These observations are leveraged by the authors to argue HOT loci as potential sites of transcriptional condensates, a claim that they provide information in support of. Transcriptional condensates are an important &quot;family&quot; of condensates, regulating different types of genes and this work supports the hypothesis that they possess similar protein partner molecules as those thought to define such bodies.</p></body></sub-article><sub-article article-type="author-comment" id="sa4"><front-stub><article-id pub-id-type="doi">10.7554/eLife.95170.3.sa4</article-id><title-group><article-title>Author response</article-title></title-group><contrib-group><contrib contrib-type="author"><name><surname>Hudaiberdiev</surname><given-names>Sanjarbek</given-names></name><role specific-use="author">Author</role><aff><institution>National Center for Biotechnology Information</institution><addr-line><named-content content-type="city">Bethesda</named-content></addr-line><country>United States</country></aff></contrib><contrib contrib-type="author"><name><surname>Ovcharenko</surname><given-names>Ivan</given-names></name><role specific-use="author">Author</role><aff><institution>National Center for Biotechnology Information</institution><addr-line><named-content content-type="city">Bethesda</named-content></addr-line><country>United States</country></aff></contrib></contrib-group></front-stub><body><p>The following is the authors’ response to the original reviews.</p><disp-quote content-type="editor-comment"><p><bold>Reviewer #1 (Public Review):</bold></p><p>Summary:</p><p>This study explores the sequence characteristics and features of high-occupancy target (HOT) loci across the human genome. The computational analyses presented in this paper provide information into the correlation of TF binding and regulatory networks at HOT loci that were regarded as lacking sequence specificity.</p><p>By leveraging hundreds of ChIP-seq datasets from the ENCODE Project to delineate HOT loci in HepG2, K562, and H1-hESC cells, the investigators identified the regulatory significance and participation in 3D chromatin interactions of HOT loci. Subsequent exploration focused on the interaction of DNA-associated proteins (DAPs) with HOT loci using computational models. The models established that the potential formation of HOT loci is likely embedded in their DNA sequences and is significantly influenced by GC contents. Further inquiry exposed contrasting roles of HOT loci in housekeeping and tissue-specific functions spanning various cell types, with distinctions between embryonic and differentiated states, including instances of polymorphic variability. The authors conclude with a speculative model that HOT loci serve as anchors where phase-separated transcriptional condensates form. The findings presented here open avenues for future research, encouraging more exploration of the functional implications of HOT loci.</p><p>Strengths:</p><p>The concept of using computational models to define characteristics of HOT loci is refreshing and allows researchers to take a different approach to identifying potential targets. The major strengths of the study lies in the very large number of datasets analyzed, with hundreds of ChIP-seq data sets for both HepG2 and K562 cells as part of the ENCODE project. Such quantitative power allowed the authors to delve deeply into HOT loci, which were previously thought to be artifacts.</p><p>Weaknesses:</p><p>While this study contributes to our knowledge of HOT loci, there are critical weaknesses that need to be addressed. There are questions on the validity of the assumptions made for certain analyses. The speculative nature of the proposed model involving transcriptional condensates needs either further validation or be toned down. Furthermore, some apparent contradictions exist among the main conclusions, and these either need to be better explained or corrected. Lastly, several figure panels could be better explained or described in the figure legends.</p></disp-quote><p>We thank the reviewer for their valuable comments.</p><p>- We have extended the study and included a new chapter focusing on the condensate hypothesis, added more supporting evidence (including the ones suggested by the reviewer), and made explicit statements on the speculative nature of this model.</p><p>- We have restructured the text to remove the sentences which might be construed as contradictory.</p><disp-quote content-type="editor-comment"><p><bold>Reviewer #2 (Public Review):</bold></p><p>Summary:</p><p>The paper 'Sequence characteristic and an accurate model of abundant hyperactive loci in human genome' by Hydaiberdiev and Ovcharenko offers comprehensive analyses and insights about the 'high-occupancy target' (HOT) loci in the human genome. These are considered genomic regions that overlap with transcription factor binding sites. The authors provided very comprehensive analyses of the TF composition characteristics of these HOT loci. They showed that these HOT loci tend to overlap with annotated promoters and enhancers, GC-rich regions, open chromatin signals, and highly conserved regions, and that these loci are also enriched with potentially causal variants with different traits.</p><p>Strengths:</p><p>Overall, the HOT loci' definition is clear and the data of HOT regions across the genome can be a useful dataset for studies that use HepG2 or K562 as a model. I appreciate the authors' efforts in presenting many analyses and plots backing up each statement.</p><p>Weaknesses:</p><p>It is noteworthy that the HOT concept and their signature characteristics as being highly functional regions of the genome are not presented for the first time here. Additionally, I find the main manuscript, though very comprehensive, long-winded and can be put in a shorter, more digestible format without sacrificing scientific content.</p><p>The introduction's mention of the blacklisted region can be rather misleading because when I read it, I was anticipating that we are uncovering new regulatory regions within the blacklisted region. However, the paper does not seem to address the question of whether the HOT regions overlap, if any, with the ENCODE blacklisted regions afterward. This plays into the central assessment that this manuscript is long-winded.</p><p>The introduction also mentioned that HOT regions correspond to 'genomic regions that seemingly get bound by a large number of TFs with no apparent DNA sequence specificity' (this point of 'no sequence specificity' is reiterated in the discussion lines 485-486). However, later on in the paper, the authors also presented models such as convolutional neural networks that take in one-hot-encoded DNA sequence to predict HOT performed really well. It means that the sequence contexts with potential motifs can still play a role in forming the HOT loci. At the same time, lines 59-60 also cited studies that &quot;detected putative drive motifs at the core segments of the HOT loci&quot;. The authors should edit the manuscript to clarify (or eradicate) contradictory statements.</p></disp-quote><p>We thank the reviewer for their valuable comments. Below are our responses to each paragraph in the given order:</p><p>We added a statement in the commenting and summarizing other publications that studied the functional aspects of HOT loci with the following sentence in the introduction part:</p><p>“Other studies have concluded that these regions are highly functionally consequential regions enriched in epigenetic signals of active regulatory elements such as histone modification regions and high chromatin accessibility”.</p><p>We significantly shortened the manuscript by (a) moving the detailed analyses of the computational model to the supplemental materials, and (b) shortening the discussions by around half, focusing on core analyses that would be most beneficial to the field.</p><p>Given that the ENCODE blacklisted regions are the regions that are recommended by the ENCODE guidelines to be avoided in mapping the ChIP-seq (and other NGS), we excluded them from our analyzed regions before mapping to the genome. Instead, we relied on the conclusions of other publications on HOT loci that the initial assessments of a fraction of HOT loci were the result of factoring in these loci which later were included in blacklisted regions.</p><p>We addressed the potential confusion by using the expression of “no sequence specificity” by (a) changing the sentence in the introduction by adding a clarification as “... with no apparent DNA sequence specificity in terms of detectible binding motifs of corresponding motifs” and (b) removing that part from the sentence in the discussions.</p><disp-quote content-type="editor-comment"><p><bold>Reviewer #3 (Public Review):</bold></p><p>Summary:</p><p>Hudaiberdiev and Ovcharenko investigate regions within the genome where a high abundance of DNA-associated proteins are located and identify DNA sequence features enriched in these regions, their conservation in evolution, and variation in disease. Using ChIP-seq binding profiles of over 1,000 proteins in three human cell lines (HepG2, K562, and H1) as a data source they're able to identify nearly 44,000 high-occupancy target loci (HOT) that form at promoter and enhancer regions, thus suggesting these HOT loci regulate housekeeping and cell identity genes. Their primary investigative tool is HepG2 cells, but they employ K562 and H1 cells as tools to validate these assertions in other human cell types. Their analyses use RNA pol II signal, super-enhancer, regular-enhancer, and epigenetic marks to support the identification of these regions. The work is notable, in that it identifies a set of proteins that are invariantly associated with high-occupancy enhancers and promoters and argues for the integration of these molecules at different genomic loci. These observations are leveraged by the authors to argue HOT loci as potential sites of transcriptional condensates, a claim that they are well poised to provide information in support of. This work would benefit from refinement and some additional work to support the claims.</p><p>Comments:</p><p>(1) Condensates are thought to be scaffolded by one or more proteins or RNA molecules that are associated together to induce phase separation. The authors can readily provide from their analysis a check of whether HOT loci exist within different condensate compartments (or a marker for them). Generally, ChIPSeq signal from MED1 and Ronin (THAP11) would be anticipated to correspond with transcriptional condensates of different flavors, other coactivator proteins (e.g., BRD4), would be useful to include as well. Similarly, condensate scaffolding proteins of facultative and constitutive heterochromatin (HP1a and EZH2/1) would augment the authors' model by providing further evidence that HOT Loci occur at transcriptional condensates and not heterochromatin condensates. Sites of splicing might be informative as well, splicing condensates (or nuclear speckles) are scaffolded by SRRM/SON, which is probably not in their data set, but members of the serine arginine-rich splicing factor family of proteins can serve as a proxy-SRSF2 is the best studied of this set. This would provide a significant improvement to their proposed model and be expected since the authors note that these proteins occur at the enhancers and promoter regions of highly expressed genes.</p><p>(2) It is curious that MAX is found to be highly enriched without its binding partner Myc, is Myc's signal simply lower in abundance, or is it absent from HOT loci? How could it be possible that a pair of proteins, which bind DNA as a heterodimer are found in HOT loci without invoking a condensate model to interpret the results?</p><p>(3) Numerous studies have linked the physical properties of transcription factor proteins to their role in the genome. The authors here provide a limited analysis of the proteins found at different HOT-loci by employing go terms. Is there evidence for specific types of structural motifs, disordered motifs, or related properties of these proteins present in specific loci?</p><p>(4) Condensates themselves possess different emergent properties, but it is a product of the proteins and RNAs that concentrate in them and not a result of any one specific function (condensates can have multiple functions!)</p><p>(5) Transcriptional condensates serve as functional bodies. The notion the authors present in their discussion is not held by practitioners of condensate science, in that condensates exist to perform biochemical functions and are dissolved in response to satisfying that need, not that they serve simply as reservoirs of active molecules. For example, transcriptional condensates form at enhancers or promoters that concentrate factors involved in the activation and expression of that gene and are subsequently dissolved in response to a regulatory signal (in transcription this can be the nascently synthesized RNA itself or other factors). The association reactions driving the formation of active biochemical machinery within condensates are materially changed, as are the kinetics of assembly. It is unnecessary and inaccurate to qualify transcriptional condensates as depots for transcriptional machinery.</p><p>1. This work has the potential to advance the field forward by providing a detailed perspective on what proteins are located in what regions of the genome. Publication of this information alongside the manuscript would advance the field materially.</p></disp-quote><p>We thank the reviewer for constructive comments and suggestions. Below are our point-by-point responses:</p><p>(1) We added a new short section “Transcriptional condensates as a model for explaining the HOT regions” with additional support for the condensate hypothesis, wherein some of the points raised here were addressed. Specifically, we used a curated LLPS proteins (CD-CODE) database and provided statistics of those annotation condensate-related DAPs.</p><p>Regarding the DAPs mentioned in this question, we observed that the distributions corresponding ChIP-seq peaks confirm the patterns expected by the reviewer (Author response image 1). Namely:</p><p>- MED1 and Ronin (THAP11) are abundant in the HOT loci, being present 67% and 64% of HOT loci respectively.</p><p>- While the BRD4 is present in 28% of the HOT loci, we observed that the DAPs with annotated LLPS activity ranged from 3% to 73%, providing further support for the condensate hypothesis.</p><p>- ENCODE database does not contain ChIP-seq dataset for HP1A. EZH2 peaks were absent in the HOT loci (0.4% overlap), suggesting the lack of heterochromatin condensate involvement.</p><p>- Serine-rich splicing factor family proteins were present only in 7.7% of the HOT loci, suggesting the absence or limited overlap with splicing condensates or nuclear speckles.</p><fig id="sa4fig1" position="float"><label>Author response image 1.</label><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-95170-sa4-fig1-v1.tif"/></fig><p>(2) In this study we selected the TF ChIP-seq datasets with stringent quality metrics, excluding those which had attached audit warning and errors. As a result, the set of DAPs analyzed in HepG2 did not include MYC, since the corresponding ChIP-seq dataset had the audit warning tags of &quot;borderline replicate concordance, insufficient read length, insufficient read depth, extremely low read depth&quot;. Analyses in K562 and H1 did include MYC (alongside MAX) ChIP-seq dataset.</p><p>To address this question, we added the mentioned ChIP-seq dataset (ENCODE ID: ENCFF800JFG) and analyzed the colocalization patterns of MYC and MAX. We observed that the MYC ChIP-seq peaks in HepG2 display spurious results, overlapping with only 5% of HOT loci. Meanwhile in K562 and H1, MYC and MAX are jointly present in 54% and 44% of the HOT loci, respectively (Author response image 2).</p><fig id="sa4fig2" position="float"><label>Author response image 2.</label><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-95170-sa4-fig2-v1.tif"/></fig><p>These observations were also supported by Jaccard indices between the MYC and MAX ChIP-seq peaks. To do this analysis, we calculated the pairwise Jaccard indices between MYC and MAX and divided them by the average Jaccard indices of 2000 randomly selected DAP pairs. In K562 and H1, the Jaccard indices between MYC and MAX are 5.72x and 2.53x greater than the random background, respectively. For HepG2, the ratio was 0.21x, clearly indicating that HepG2 MYC ChIP-seq dataset is likely erroneous.</p><fig id="sa4fig3" position="float"><label>Author response image 3.</label><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-95170-sa4-fig3-v1.tif"/></fig><p>(3) Despite numerous publications focusing on different structural domains in transcription factors, we could not find an extensive database or a survey study focusing on annotations of structural motifs in human TFs. Therefore, surveying such a scale would be outside of this study’s scope. We added only the analysis of intrinsically disordered regions, as it pertains to the condensate hypothesis. To emphasize this shortcoming, we added the following sentence to the end of the discussions section.</p><p>“Further, one of the hallmarks of LLPS proteins that have been associated with their abilities to phase-separate is the overrepresentation of certain structural motifs, which we did not pursue due to size limitations.”</p><p>(4, 5) We agree with these statements and thank the reviewer for pointing out this faulty statement. We modified the sections in the discussions related to the condensates and removed the part where we implied that the condensate model could be because of mostly a single function of TF reservoir.</p><p>(6) We added a table to the supplemental materials (Zenodo repository) with detailed annotation of HOT and non-HOT DAP-bound loci in the genome.</p><disp-quote content-type="editor-comment"><p><bold>Recommendations for the authors:</bold></p><p><bold>Reviewing Editor (Recommendations For The Authors):</bold></p><p>The clause with &quot;inadequate&quot; would be dropped if the authors sufficiently address reviewer concerns about clarity of writing, including:</p><p>(1) Editing the title to better reflect the findings of the paper.</p><p>(2) Making clear that the condensate model is speculative and not explicitly tested in this study (and may be better described as a hypothesis).</p><p>(3) Resolving apparent contradictions regarding DNA sequence specificity and the interpretation of ChIP-seq signal intensity.</p><p>(4) Better specifying and justifying model parameters, thresholds, and assumptions.</p><p>(5) Shortening the manuscript to emphasize the main, well-supported claims and to enhance readability (especially the discussion section).</p></disp-quote><p>We thank the Editor for their work. We followed their advice and implemented changes and additions to address all 5 points.</p><disp-quote content-type="editor-comment"><p><bold>Reviewer #1 (Recommendations For The Authors):</bold></p><p>(1) The title &quot;Sequence characteristics and an accurate model of abundant hyperactive loci in the human genome&quot; does not accurately reflect the findings of the paper. We are unclear as to what the 'accurate model' refers to. Is it the proposed model 'based on the existence of large transcriptional condensates' (abstract)? If so, there are concerns below regarding this statement (see comment 2). If the authors are referring to the computational modeling presented in Figure 5, it is unclear that any one of them performed that much better than the others and the best single model was not identified. Furthermore, the models being developed in the study constitute only a portion of the paper and lacked validation through additional datasets. Additionally, sequence characteristics were not a primary focus of the study. Only figure 5 talks about the model and sequence characteristics, the rest of the figures are left out of the equation.</p></disp-quote><p>We agree with and thank the reviewer for this idea of clarifying the intended meaning.</p><p>(1) We changed the title and clarified that the computational model is meant:</p><p>“Functional characteristics and a computational model of abundant hyperactive loci in the human genome”.</p><p>(2) Shortened the part of the manuscript discussing the computational models and pointed out the CNNs as “the best single model”.</p><disp-quote content-type="editor-comment"><p>(2) The abstract and discussion (and perhaps the title) propose a model of transcriptional condensates in relation to HOT loci. However, there is no data provided in the manuscript that relates to condensates. Therefore, anything relating to condensates is primarily speculative. This distinction needs to be properly made, especially in the abstract (and cannot be included in the title). Otherwise, these statements are misleading. Although the field of transcriptional condensates is relatively new, there have been several factors studied. The authors could include in Figure 2d which factors have been shown to form transcriptional condensates. This might provide some support for the model, though it would still largely remain speculative unless further testing is done.</p></disp-quote><p>We added a new short chapter “Transcriptional condensates as a model for explaining the HOT regions”, with additional analyses testing the condensates hypothesis. We provided supportive evidence by analyzing the metrics used as hallmarks of condensates including the distributions of annotated condensate-related proteins, nascent transcription, and protein-RNA interaction levels in HOT loci. Still, we acknowledge that this is a speculative hypothesis and we clarified that with the following statement in the discussions:</p><p>“It is important to note here that our proposed condensate model is a speculative hypothesis. Further experimental studies in the field are needed to confirm or reject it.”</p><disp-quote content-type="editor-comment"><p>(3) Several apparent contradictions exist throughout the manuscript. For example, &quot;HOT locus formation are likely encoded in their DNA sequences&quot; (lines 329-330) vs the proposed model of formation through condensates (abstract). These two statements do not seem compatible, or at the very least, the authors can explain how they are consistent with each other. Another example: &quot;ChIP-seq signal intensity as a proxy for... binding affinity&quot; (line 229) vs. &quot;ChIP-seq signal intensities do not seem to be a function of the DNA-binding properties of the DAPs&quot; (lines 259-260). The first statement is the assumption for subsequent analyses, which has its own concerns (see comment 4). But the conclusion from that analysis seems to contradict the assumption, at least as it is stated.</p></disp-quote><p>In this study, we argue that the two statements may not necessarily contradict each other. We aimed to (a) demonstrate that the observed intensity of DAP-DNA interactions as measured by ChIP-seq experiments at HOT loci cannot be explained with direct DNA-binding events of the DAPs alone and (b) propose a hypothesis that this observation can be at least partially explained if the HOT loci have the propensity to either facilitate or take part in the formation of transcriptional condensates.</p><p>One of the conditions for condensates to form at enhancers was shown to be the presence of strong binding sites of key TFs (Shrinivas et al. 2019 “Enhancer features that drive the formation of transcriptional condensates”), where the study was conducted using only one TF (OCT4) and one coactivator (MED1). To the best of our knowledge, no such study has been conducted involving many TFs and cofactors simultaneously. We also know that the factors that lead to liquid-to-liquid phase separation include weak multivalent IDR-IDR, IDR-DNA, and IDR-RNA interactions. As a result, the observed total sum of ChIP-seq peaks in HOT loci is the direct DNA-binding events combined with the indirect DAP-DNA interactions, some of which may be facilitated by condensates. And, the fact that CNNs can recognize the HOT loci with high accuracy suggests that there must be an underlying motif grammar specific to HOT loci.</p><p>We emphasized this conclusion in the discussions.</p><p>The comment on using the ChIP-seq signal as a proxy for DNA-binding affinity is addressed under comment 4.</p><disp-quote content-type="editor-comment"><p>(4) In lines 229-230, the authors used &quot;the ChIP-seq signal intensity as a proxy for the DAP binding affinity.&quot; What is the basis for this assumption? If there is a study that can be referenced, it should be added. However, ChIP-seq signal intensity is generally regarded as a combination of abundance, frequency, or percentage of cells with binding. RNA Pol2 is a good example of this as it has no specific binding affinity but the peak heights indicate level of expression. Therefore, the analyses and conclusions in Figure 4, particularly panel A, are problematic. In addition, clarification from lines 258-260 is needed as it contradicts the earlier premise of the section (see comment 3).</p></disp-quote><p>We thank the reviewer for pointing out this error. The main conclusion of the paragraph is that the average ChIP-seq signal values at HOT loci do not correlate well with the sequence-specificity of TFs. We reworded the paragraph stating that we are analyzing the patterns of ChIP-seq signals across the HOT loci, removing the part that we use them as a proxy for sequence-specific binding affinity.</p><disp-quote content-type="editor-comment"><p>(5) In Figure 1A, the authors show that &quot;the distribution of the number of loci is not multimodal, but rather follows a uniform spectrum, and thus, this definition of HOT loci is ad-hoc&quot; (lines 92-95). The threshold to determine how a locus is considered to be HOT is unclear. How did the authors decide to use the current threshold given the uniform spectrum observed? How does this method of calling HOT loci compare to previous studies? How much overlap is there in the HOT loci in this study versus previous ones?</p></disp-quote><p>We moved the corresponding explanation from the supplemental methods to the main methods section of the manuscript.</p><p>Briefly, our reasoning was as follows: assuming that an average TFBS is 8bp long and given that we analyze the loci of length 400bp, we can set the theoretical maximum number of simultaneous binding events to be 50. Hence, if there are &gt;50 TF ChIP-seq peaks in a given 400bp locus, it is highly unlikely that the majority of ChIP-seq peaks can be explained by direct TF-DNA interactions. The condition of &gt;50 TFs corresponded to the last four bins of our binning scale, which was used as an operational definition for HOT loci.</p><p>We have compared our definition of HOT loci to those reported in previous studies by Remaker et al. and Boyle et al. The results of our analyses are in lines 147-154.</p><disp-quote content-type="editor-comment"><p>(6) In Figure 3B, the authors state that of &quot;the loop anchor regions with &gt;3 overlapping loops, 51% contained at least one HOT locus, suggesting an interplay between chromatin loops and HOT loci.&quot; However, it is unclear how &quot;51%&quot; is calculated from the figure. Similarly, in the following sentence, &quot;94% of HOT loci are located in regions with at least one chromatin interaction&quot;. It is unclear as to how the number was obtained based on the referenced figure.</p></disp-quote><p>Initially, the x-axis on the Figure 3B was missing, making it hard to understand what we meant. We added the x-axis numbers and changed the “51%” to “more than half”. We intend to say that, of the loci with 4 and 5 overlapping loops, exactly 50% contain at least one HOT locus. However, since for x=6 the percentage is 100% (since there’s only one such locus), the percentage is technically “more than half”.</p><p>The percentage of HOT loci engaging in chromatin interaction regions (91%) was calculated by simply overlapping the HOT regions with Hi-C long-range contact anchors. The details of extracting these regions using FitHiChip are described in Supplemental Methods 1.3.</p><disp-quote content-type="editor-comment"><p>(7) While we have a limited basis to evaluate computational models, we would like to see a clearer explanation of the model set-up in terms of the number of trained vs. test datasets. In addition, it would be interesting to see if the models can be applied to data from different cell lines.</p></disp-quote><p>We added the table with the sizes of the datasets used for classification in Supplemental Methods 1.6.1.</p><p>Evaluating the models trained on the HOT loci of HepG2 and K562 on other cell lines would pose challenges since the number of available ENCODE TF ChIP-seq datasets is significantly less compared to the mentioned cell lines. Therefore, we conducted the proposed analysis between the studied cell lines. Specifically, we used the CNN models trained on HOT and regular enhancers of HepG2 and K562. Then, we evaluated each model on the test sets of each classification experiment (Author response image 4). We observed that the classification results of the HOT loci demonstrated a higher level of tissue-specificity compared to the same classification results of the regular enhancers.</p><fig id="sa4fig4" position="float"><label>Author response image 4.</label><graphic mimetype="image" mime-subtype="tiff" xlink:href="elife-95170-sa4-fig4-v1.tif"/></fig><disp-quote content-type="editor-comment"><p>(8) Lines 349-351. The significance of highly expressed genes being more prone to having multiple HOT loci, and vice versa, appears conventional and remains unclear. Intuitively, it makes sense for higher expressed genes to have more of the transcriptional machinery bound, and would bias the analysis. One way to circumvent this is to only analyze sequence-specific TFs and remove ones that are directly related to transcription machinery.</p></disp-quote><p>We thank the reviewer for this suggestion. Our attempt to re-annotate the HOT loci with only sequence-specific TFs led to a significantly different set of loci, which would not be strictly comparable to the HOT loci defined by this study. Analyzing these new sets of loci would create a noticeable departure from the flow of the manuscript and further extend the already long scope of the study.</p><p>Moreover, numerous studies have shown that super-enhancers recruit large numbers of TFs via transcriptional condensates (Boija et al., 2018; Cho et al., 2018; Sabari et al., 2018). We hope that our results can serve as data-driven supportive evidence for those studies.</p><disp-quote content-type="editor-comment"><p>(9) Lines 393-396. We would like to see a reference to the models shown in the figures, if these models have been published previously.</p></disp-quote><p>We could not understand the question. The lines 393-396 contains the following sentence:</p><p>“However, many of the features of the loci that we’ve analyzed so far demonstrated similar patterns (GC contents, target gene expressions, ChIP-seq signal values etc.) when compared to the DAP-bound loci in HepG2 and K562, suggesting that albeit limited, the distribution of the DAPs in H1 likely reflects the true distribution of HOT loci.”</p><p>In case the question was about the models that we trained to classify the HOT loci, we included the models and codebase to Zenodo and GitHub repository.</p><disp-quote content-type="editor-comment"><p>(10) Values in Figure 7D are not reflected in the text. Specifically, the text states &quot;Average ... phastCons of the developmental HOT loci are 1.3x higher than K562 and HepG2 HOT loci (Figure 7D)&quot; (lines 408-409). Figure 7D shows conservation scores between HOT enhancers vs promoters for each cell line, and does not seem to reflect the text.</p></disp-quote><p>We modified the figure to reflect the statement appropriately.</p><disp-quote content-type="editor-comment"><p>(11) Methodology should include a justification for the use of the Mann-Whitney U-test (non-parametric) over other statistical tests.</p></disp-quote><p>We added the following description to the methods section:</p><p>“For calculating the statistical significance, we used the non-parametric Mann-Whitney U-test when the compared data points are non-linearly correlated and multi-modal. When the data distributions are bell-curve shaped, the Student’s t-test was used.“</p><disp-quote content-type="editor-comment"><p>Minor:</p><p>(1) Figure 2b was never mentioned in the paper. This can be added alongside Figure S6C, line 148.</p></disp-quote><p>Indeed, Figure 2B was supposed to be listed together with Figure S6C, which was omitted by mistake. It was corrected.</p><disp-quote content-type="editor-comment"><p>(2) Supplementary Figure 8 has two Cs. Needs to be corrected to D.</p></disp-quote><p>Fixed.</p><disp-quote content-type="editor-comment"><p>(3) Figure 3B is missing labels on the x-axis.</p></disp-quote><p>Fixed.</p><disp-quote content-type="editor-comment"><p>(4) The horizontal bar graph on the bottom left of Figure 1E needs to be described in the figure legend.</p></disp-quote><p>Description added to the figure caption.</p><disp-quote content-type="editor-comment"><p>(5) Line 345, Fig 15A should be Fig S15A.</p></disp-quote><p>Corrected.</p><disp-quote content-type="editor-comment"><p><bold>Reviewer #2 (Recommendations For The Authors):</bold></p><p>I listed all my concerns about the paper in the public comments. I think the manuscript is very comprehensive and it is valuable, but it should be cut short and presented in a more digestible way.</p></disp-quote><p>We thank the reviewer for their valuable comments and suggestions. We addressed all the concerns listed in the public comments. We shortened the manuscript by reducing the paragraph that focuses on computational classification models and reduced the discussions by about half in length.</p><disp-quote content-type="editor-comment"><p>Line 55: What are chromatin-associated proteins, i.e. are they histone modifications?</p></disp-quote><p>To clarify the definition used from the citation we changed the sentence to the following:</p><p>“For instance, Partridge et al. studied the HOT loci in the context of 208 proteins including TFs, cofactors, and chromatin regulators which they called chromatin-associated proteins.”</p><disp-quote content-type="editor-comment"><p>Though most of the paper can be cut short to avoid analysis paralysis for readers, there are details that still need filling in. For example, how did the authors perform PCA analysis, i.e. what are the features of each data point in the PCA analysis? Lines 214-215: How do we calculate the number of multi-way contacts in Hi-C data?</p></disp-quote><p>We added clarifying descriptions and changed the mentioned sentences to the following:</p><p>PCA:</p><p>“To analyze the signatures of unique DAPs in HOT loci, we performed a PCA analysis where each HOT locus is represented by a binary (presence/absence) vector of length equal to the total number of DAPs analyzed.”</p><p>Multi-way contacts on loop anchors:</p><p>“To investigate further, we analyzed the loop anchor regions harboring HOT loci and observed that the number of multi-way contacts on loop anchors (i.e. loci which serve as anchors to multiple loops) correlates with the number of bound DAPs (rho=0.84 p-value&lt;10E-4; Pearson correlation). “</p><disp-quote content-type="editor-comment"><p>- Lines 251-252: How did the referenced study categorize DAPs? It is important for any manuscript to be self-contained.</p></disp-quote><p>We added the explanation and changed the sentence to the following:</p><p>“To test this hypothesis, we classified the DAPs into those two categories using the definitions provided in the study (Lambert et al. 2018) 28, where the TFs are classified by manual curation through extensive literature review and supported by annotations such as the presence of DNA-binding domains and validated binding motifs<bold>.</bold> Based on this classification, we categorized the ChIP-seq signal values into these two groups.“</p><disp-quote content-type="editor-comment"><p>- Lines 181-185, sentences starting with 'To test' can be moved to the methods, leaving only brief mentions of the statistic tests if needed.</p></disp-quote><p>We removed the mentioned sentence and moved to the supplemental methods (1.4).</p><disp-quote content-type="editor-comment"><p>- Lines 217-220: I find this sentence extremely redundant unless it can offer more specific insights about a particular set of DAPs or if the DAPs are closer/or a proven distal enhancer to a confirmed causal gene.</p></disp-quote><p>We removed the mentioned sentence from the text.</p><disp-quote content-type="editor-comment"><p>- Lines 243-246: How did the authors determine the set DAPs that have stabilizing effects, and how exactly are the 'stabilizing effects' observed/measured?</p></disp-quote><p>We added explanations to Supplemental Methods 3.1 and Fig S18, S19.</p><p>While addressing this comment we realized that the reported value of the ratio is 1.91x, not 1.7x. We corrected that value in the main text and added the p-value.</p><disp-quote content-type="editor-comment"><p>- When discussing the phastCons scores analyses, such as in lines 268-271, how did the authors calculate the relationship between phastCons scores and HOT loci, i.e. was the score averaged across the 400-bp locus to obtain a locus-specific conservation score?</p></disp-quote><p>Yes, per-locus conservation scores were averaged over the bps of loci. We added this clarification to the methods.</p><disp-quote content-type="editor-comment"><p>- Line 311: What is the role of the 'control sets' in the analyses of the sequence's relationship with HOT?</p></disp-quote><p>In this specific case, the control sets are used as background or negative sets to set up the classification tasks. In other words, we are asking, whether the HOT loci can be distinguished when compared to random chromatin-accessible regions, promoters, or regular enhancers. We clarified this in the text.</p><disp-quote content-type="editor-comment"><p>- I also find the discussion about different machine learning methods that classify HOT loci based on sequence contexts quite redundant UNLESS the authors decide to go further into the features' importance (such as motifs) in the models that predict/ are associated with HOT loci, which in itself can constitute another study.</p></disp-quote><p>We agree with the reviewer, and shortened the part with the discussions of models by limiting it to only 3 main models and moved the rest to the supplemental materials.</p><disp-quote content-type="editor-comment"><p>- Can the authors clarify where they obtain data on super-enhancers?</p></disp-quote><p>We obtained the super-enhancer definitions from the original study (Hnisz et al. 2013, PMID: 24119843) where the super-enhancers were defined for multiple cell lines. We clarified this in the methods.</p><disp-quote content-type="editor-comment"><p>- Figure 1B, the x and y axis should be clarified.</p></disp-quote><p>We clarified it by using MAX as an example case in the figure caption as follows:</p><p>“Prevalence of DAPs in HOT loci. Each dot represents a DAP. X-axis: percentage of HOT loci in which DAP is present (e.g. MAX is present in 80% of HOT loci). Y-axis: percentage of total peaks of DAPs that are located in HOT loci (e.g. 45% of all the ChIP-seq peaks of MAX is located in the HOT loci). Dot color and size are proportional to the total number of ChIP-seq peaks of DAP.”</p><disp-quote content-type="editor-comment"><p><bold>Reviewer #3 (Recommendations For The Authors):</bold></p><p>The list of proteins associated with different types of genomic loci at a meta level (enhancers, promoters, and gene body etc.), and an annotation of the genome at the specific loci level.</p><p>The authors use a wide range of acronyms throughout the text and figure legends, they do a reasonably good job, but the main text section &quot;HOT-loci are enriched in causal variants&quot; and Figure 8 would be materially improved if they held it to the same standard.</p><p>Size is a physical property and not a physicochemical property.</p></disp-quote><p>We thank the reviewer for their comments and suggestions. We added a table to supplemental files with detailed annotations of analyzed loci.</p><p>We reviewed the section “HOT loci are enriched in causal variants” and corrected a few mismatches in the acronyms.</p></body></sub-article></article>